# A Gallium driver for the Intel Poulsbo GMA500 (PowerVR SGX535). # # Adds src/gallium/drivers/sgx and hooks it into the build and the DRM loader. # The part is a tile-based deferred renderer with no fixed-function transform, # blend or clear: everything is USSE code the driver generates, so the driver # carries its own shader compiler (TGSI in, USSE out) and builds the frame's # parameter heap by hand. # # The kernel half is gma500-sgx-render.patch in the linux package; the two # share sgx_drm.h as their interface and have to be updated together. Without # the kernel side there is no render node and the loader falls back to # llvmpipe. # # The Mesa loader matches this device as "gma500", not "sgx"; the alias is in # the driver descriptor and both names resolve to the same screen. # # This only adds 'sgx' to the gallium-drivers choices - it does not enable it. # To build it, append it to $galdrv in mesa.desc for the x86 32-bit case. # # Written from hardware measurement against the vendor's own DDK. # Signed-off-by: Rene Rebe diff -urNp mesa-26.2.2.orig/meson.build mesa-26.2.2/meson.build --- mesa-26.2.2.orig/meson.build 2026-09-02 17:40:02.000000000 +0200 +++ mesa-26.2.2/meson.build 2026-09-08 10:57:36.694119303 +0200 @@ -228,6 +228,7 @@ with_gallium_tegra = gallium_drivers.con with_gallium_crocus = gallium_drivers.contains('crocus') with_gallium_iris = gallium_drivers.contains('iris') with_gallium_i915 = gallium_drivers.contains('i915') +with_gallium_sgx = gallium_drivers.contains('sgx') with_gallium_svga = gallium_drivers.contains('svga') with_gallium_virgl = gallium_drivers.contains('virgl') with_gallium_lima = gallium_drivers.contains('lima') diff -urNp mesa-26.2.2.orig/meson.options mesa-26.2.2/meson.options --- mesa-26.2.2.orig/meson.options 2026-09-02 17:40:02.000000000 +0200 +++ mesa-26.2.2/meson.options 2026-09-08 10:57:36.694178576 +0200 @@ -86,6 +86,7 @@ option( choices : [ 'all', 'auto', 'asahi', 'crocus', 'd3d12', 'ethosu', 'etnaviv', 'freedreno', 'i915', 'iris', + 'sgx', 'lima', 'llvmpipe', 'nouveau', 'panfrost', 'r300', 'r600', 'radeonsi', 'rocket', 'softpipe', 'svga', 'tegra', 'v3d', 'vc4', 'virgl', 'zink', ], diff -urNp mesa-26.2.2.orig/src/gallium/auxiliary/pipe-loader/pipe_loader_drm.c mesa-26.2.2/src/gallium/auxiliary/pipe-loader/pipe_loader_drm.c --- mesa-26.2.2.orig/src/gallium/auxiliary/pipe-loader/pipe_loader_drm.c 2026-09-02 17:40:02.000000000 +0200 +++ mesa-26.2.2/src/gallium/auxiliary/pipe-loader/pipe_loader_drm.c 2026-09-08 10:57:36.694351135 +0200 @@ -68,6 +68,8 @@ static const struct pipe_loader_ops pipe static const struct drm_driver_descriptor *driver_descriptors[] = { &i915_driver_descriptor, + &sgx_driver_descriptor, + &gma500_driver_descriptor, &iris_driver_descriptor, &crocus_driver_descriptor, &nouveau_driver_descriptor, diff -urNp mesa-26.2.2.orig/src/gallium/auxiliary/target-helpers/drm_helper.h mesa-26.2.2/src/gallium/auxiliary/target-helpers/drm_helper.h --- mesa-26.2.2.orig/src/gallium/auxiliary/target-helpers/drm_helper.h 2026-09-02 17:40:02.000000000 +0200 +++ mesa-26.2.2/src/gallium/auxiliary/target-helpers/drm_helper.h 2026-09-08 10:57:36.694273671 +0200 @@ -55,6 +55,25 @@ const struct drm_driver_descriptor descr #undef GALLIUM_ETHOSU #endif + +#ifdef GALLIUM_SGX +#include "sgx/sgx_public.h" + +static struct pipe_screen * +pipe_sgx_create_screen(int fd, const struct pipe_screen_config *config) +{ + struct pipe_screen *screen; + + (void)config; + screen = sgx_screen_create(fd); + return screen ? debug_screen_wrap(screen) : NULL; +} +DRM_DRIVER_DESCRIPTOR(sgx, NULL, 0) +DRM_DRIVER_DESCRIPTOR_ALIAS(sgx, gma500, NULL, 0) +#else +DRM_DRIVER_DESCRIPTOR_STUB(sgx) +DRM_DRIVER_DESCRIPTOR_STUB(gma500) +#endif #ifdef GALLIUM_I915 #include "i915/drm/i915_drm_public.h" #include "i915/i915_public.h" diff -urNp mesa-26.2.2.orig/src/gallium/auxiliary/target-helpers/drm_helper_public.h mesa-26.2.2/src/gallium/auxiliary/target-helpers/drm_helper_public.h --- mesa-26.2.2.orig/src/gallium/auxiliary/target-helpers/drm_helper_public.h 2026-09-02 17:40:02.000000000 +0200 +++ mesa-26.2.2/src/gallium/auxiliary/target-helpers/drm_helper_public.h 2026-09-08 10:57:36.694303242 +0200 @@ -5,6 +5,8 @@ struct pipe_screen; struct pipe_screen_config; extern const struct drm_driver_descriptor i915_driver_descriptor; +extern const struct drm_driver_descriptor sgx_driver_descriptor; +extern const struct drm_driver_descriptor gma500_driver_descriptor; extern const struct drm_driver_descriptor iris_driver_descriptor; extern const struct drm_driver_descriptor crocus_driver_descriptor; extern const struct drm_driver_descriptor nouveau_driver_descriptor; diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/Xpsb.h mesa-26.2.2/src/gallium/drivers/sgx/Xpsb.h --- mesa-26.2.2.orig/src/gallium/drivers/sgx/Xpsb.h 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/Xpsb.h 2026-09-08 10:57:36.684489238 +0200 @@ -0,0 +1,87 @@ +/* Public API of the Xpsb Xorg sub-module. + * + * This is the contract xserver-xorg-video-psb links against. The layout of + * XpsbSurface and the prototypes below must match the original header + * (src/Xpsb.h in the DDX source, Copyright (c) Intel Corp. 2007, written by + * Thomas Hellstrom at Tungsten Graphics, MIT licensed) or the DDX will pass + * malformed arguments. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#ifndef _XPSB_H_ +#define _XPSB_H_ + +#ifdef HAVE_XORG_CONFIG_H +#include +#endif +#include "xorg-server.h" +#include "xf86drm.h" +#include "xf86mm.h" +#include "xf86.h" +#include "xf86_OSproc.h" + +#define XPSB_VOFFSET_X 0 +#define XPSB_VOFFSET_Y 1 +#define XPSB_VOFFSET_UT0 2 +#define XPSB_VOFFSET_VT0 3 +#define XPSB_VOFFSET_UT1 4 +#define XPSB_VOFFSET_VT1 5 +#define XPSB_VOFFSET_UT2 6 +#define XPSB_VOFFSET_VT2 7 +#define XPSB_VOFFSET_COUNT 8 + +typedef enum +{ Xpsb_nearest = 0, Xpsb_linear } +XpsbFilterFormats; +typedef enum +{ Xpsb_repeat = 0, Xpsb_clamp, Xpsb_clampGL } +XpsbAddrModes; + +typedef struct _XpsbSurface +{ + drmBO *buffer; + unsigned int offset; + unsigned int pictFormat; + unsigned int w; + unsigned int h; + unsigned int stride; + + XpsbFilterFormats minFilter; + XpsbFilterFormats magFilter; + XpsbAddrModes uMode; + XpsbAddrModes vMode; + unsigned int texCoordIndex; + Bool isYUVPacked; + unsigned int packedYUVId; + + unsigned int x; + unsigned int y; +} XpsbSurface, *XpsbSurfacePtr; + +extern Bool XpsbInit(ScrnInfoPtr pScrn, CARD8 * map, int drmFD); +extern void XpsbTakeDown(ScrnInfoPtr pScrn); + +extern int psb3DPrepareComposite(ScrnInfoPtr pScrn, XpsbSurfacePtr dst, + XpsbSurfacePtr opTextures[], + int numOpTextures, int compOp, + unsigned int scalar, Bool scalarSrc, + Bool scalarMask); +extern void psb3DCompositeQuad(ScrnInfoPtr pScrn, float vertices[]); +extern int psb3DCompositeFinish(ScrnInfoPtr pScrn); + +extern int psbBlitYUV(ScrnInfoPtr pScrn, XpsbSurfacePtr dst, + XpsbSurfacePtr backTextures[], int numBackTextures, + Bool isPlanar, unsigned int planarID, float texCoord0[], + float texCoord1[], float texCoord2[], int numCoord, + float conversion_data[]); +extern int psbBlitYUVDetear(ScrnInfoPtr pScrn, XpsbSurfacePtr dst, + XpsbSurfacePtr backTextures[], + int numBackTextures, Bool isPlanar, + unsigned int planarID, float texCoord0[], + float texCoord1[], float texCoord2[], + int numCoord, float conversion_data[]); +extern int XpsbCmdCancelBlit(ScrnInfoPtr pScrn); + +#endif /* _XPSB_H_ */ diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/meson.build mesa-26.2.2/src/gallium/drivers/sgx/meson.build --- mesa-26.2.2.orig/src/gallium/drivers/sgx/meson.build 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/meson.build 2026-09-08 10:57:36.681338773 +0200 @@ -0,0 +1,63 @@ +# Copyright (C) 2026 RenĂ© Rebe +# SPDX-License-Identifier: MIT + +files_sgx = files( + 'sgx_pipe.c', + 'sgx_pipe_vbuf.c', + 'sgx_screen.c', + 'sgx_context.c', + 'sgx_state.c', + 'sgx_hwtcl.c', + 'sgx_shader.c', + 'sgx_sampler.c', + 'sgx_resource.c', + 'sgx_winsys.c', + 'sgx_winsys_drm.c', + # the compiler: TGSI in, SGX535 machine code out + 'tgsi_parse.c', + 'tgsi_to_uir.c', + 'usse_ir.c', + 'usse_cg.c', + 'usse_isa.c', + 'usse-core.c', + # the frame generators, reverse engineered and exercised on hardware + 'xpsb_frame.c', + 'xpsb_heap.c', + 'xpsb_pds.c', + 'xpsb_vidshader.c', + 'xpsb_pixel.c', + 'xpsb_yuv.c', + 'xpsb_shader.c', +) + +libsgx = static_library( + 'sgx', + files_sgx, + gnu_symbol_visibility : 'hidden', + include_directories : [inc_include, inc_src, inc_gallium, inc_gallium_aux, + include_directories('.')], + dependencies : [idep_nir, idep_mesautil], + c_args : ['-DGALLIUM_SGX'], +) + +driver_sgx = declare_dependency( + compile_args : '-DGALLIUM_SGX', + link_with : [libsgx], +) + +# The vtable test, built here because sgx_pipe.c now needs the draw module and +# nir_to_tgsi: outside this build there is no honest way to compile it. It +# drives the vtables into a mock kernel, so it needs no device. +if get_option('build-tests') + test('sgx-pipe', + executable( + 'sgx_pipe_test', + files('sgx_pipe_test.c'), + include_directories : [inc_include, inc_src, inc_gallium, inc_gallium_aux, + include_directories('.')], + dependencies : [idep_nir, idep_mesautil], + link_with : [libsgx, libgallium], + ), + suite : ['sgx'], + ) +endif diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_check.h mesa-26.2.2/src/gallium/drivers/sgx/sgx_check.h --- mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_check.h 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/sgx_check.h 2026-09-08 10:57:36.681308562 +0200 @@ -0,0 +1,114 @@ +/* SPDX-License-Identifier: GPL-2.0-only */ +/* + * The two pieces of sgx_drv.c that are pure logic: which registers a submitted + * stream may write, and where a binding is allowed to land. They live in a + * header so the module and a host test can share them verbatim - the whitelist + * is a security boundary, and testing it only once the hardware is wired up + * would be leaving it untested for as long as that takes. + * + * Copyright (C) 2026 RenĂ© Rebe + */ +#ifndef _SGX_CHECK_H_ +#define _SGX_CHECK_H_ + +/* + * Register whitelist. + * + * A submitted stream is {offset, value} pairs, which is what the hardware + * consumes, but an unchecked one lets userspace write any register on the + * device. These ranges are not guessed: they are every register the frame + * generators in tools/xpsb-open actually emit, enumerated and checked by + * test/sgx_realframe_test.c so the list cannot quietly stop matching what a + * working frame needs. + * + * An earlier version of this was three round ranges and was wrong in both + * directions - it refused 0x0800, 0x0a5c-0x0a64 and 0x0cb0, which a real + * raster stream writes, and it allowed the whole DPM block. That second part + * mattered more: 0x062c is DPM_PAGE_TABLE, and pointing it somewhere wrong is + * exactly the fault the out-of-memory work spent days on. The DPM belongs to + * the kernel's scene logic and no submitted stream has any business there. + */ +static const struct { unsigned int first, last; } sgx_reg_ok[] = { + { 0x0204, 0x0250 }, /* TA: parameter base, vertex source, kick */ + /* ISP and 3D, including the background object. Split around 0x0428, + * EUR_CR_ISP_START_RENDER: a stream that pulses it starts the render + * itself, before the scene is bound, and no real frame writes it - + * the kernel's kick sequence is the only place it belongs. */ + { 0x0400, 0x0424 }, + { 0x042c, 0x04dc }, + { 0x0800, 0x0800 }, + { 0x0a5c, 0x0a64 }, + { 0x0cb0, 0x0cb0 }, /* BIF 3D flush, which the raster stream emits */ +}; + +static inline int sgx_reg_allowed(unsigned int off) +{ + unsigned int i; + + if (off & 3) + return 0; + for (i = 0; i < sizeof sgx_reg_ok / sizeof sgx_reg_ok[0]; i++) + if (off >= sgx_reg_ok[i].first && off <= sgx_reg_ok[i].last) + return 1; + return 0; +} + +/* Nonzero if the stream is not acceptable: -1 if it is not whole {offset,value} + * pairs, 1 if an offset is outside the whitelist, in which case *bad - when + * given - is the dword index of that offset. Zero means acceptable. */ +static inline int sgx_stream_bad_at(const unsigned int *s, unsigned int dwords, + unsigned int *bad) +{ + unsigned int i; + + if (dwords & 1) + return -1; /* not whole pairs */ + for (i = 0; i < dwords; i += 2) + if (!sgx_reg_allowed(s[i])) { + if (bad) + *bad = i; + return 1; + } + return 0; +} + +/* Does [va, va+size) sit inside the window? */ +static inline int sgx_in_window(unsigned long long va, unsigned long long size, + unsigned long long start, + unsigned long long wsize) +{ + if (!size) + return 0; + if (va < start) + return 0; + if (va + size < va) /* wrap */ + return 0; + return va + size <= start + wsize; +} + +/* Do two bindings overlap? */ +static inline int sgx_overlaps(unsigned long long a, unsigned long long alen, + unsigned long long b, unsigned long long blen) +{ + return a < b + blen && b < a + alen; +} + +/* Reserved fields must be zero and unknown flags must be refused. + * + * Not a style rule. A field that is ignored today cannot be given a meaning + * tomorrow: an old kernel accepts a new flag, ignores it, and does something + * other than what the caller asked - silently. Refusing what is not understood + * is what makes the interface extensible at all, and it is the first thing + * asked of a new UAPI. + */ +static inline int sgx_flags_ok(unsigned int flags, unsigned int known) +{ + return (flags & ~known) == 0; +} + +static inline int sgx_pad_ok(unsigned int pad) +{ + return pad == 0; +} + +#endif /* _SGX_CHECK_H_ */ diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_context.c mesa-26.2.2/src/gallium/drivers/sgx/sgx_context.c --- mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_context.c 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/sgx_context.c 2026-09-08 10:57:36.678794206 +0200 @@ -0,0 +1,8064 @@ +/* The driver-side context - see sgx_context.h. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#include "sgx_context.h" +#include "sgx_hwtcl.h" +#include "sgx_cookie.h" +#include "sgx_drm.h" +#include "xpsb_frame.h" +#include "xpsb_shader.h" +#include "sgx_sampler.h" + +#include +#include +#include +#include +#include +#include +#include + +/* SGX535 rev 1.2.1 enables neither workaround; see sgx_cookie.h. *//* getenv() is a linear scan of the environment, and this file asks it about + * twenty times per draw record. What it answers cannot change during a run, so + * each site remembers it. Measured on ioquake3, where the per-draw and + * per-record diagnostics were a large part of a frame's CPU time. */ +#define SGX_ENV(name) __extension__({ \ + static int sgx_env_cached_ = -1; \ + if (sgx_env_cached_ < 0) \ + sgx_env_cached_ = getenv(name) != NULL; \ + sgx_env_cached_; \ +}) + +/* The same for a knob whose value is read, not just its presence. getenv() is + * a linear scan of the environment and glibc's does a strncmp per entry, which + * measured at nearly two fifths of ioquake3's CPU once the per-record loops + * below started asking it per record and per texture unit. The answer cannot + * change during a run, so each site remembers it. */ +#define SGX_ENVS(name) __extension__({ \ + static const char *sgx_envs_cached_; \ + static int sgx_envs_got_; \ + if (!sgx_envs_got_) { \ + sgx_envs_cached_ = getenv(name); \ + sgx_envs_got_ = 1; \ + } \ + sgx_envs_cached_; \ +}) + + +/* Whether the MTE applies the viewport, rather than a shader epilogue reading + * it out of the vertex record. The vendor programs group 8 of the state block + * and clears bit 16 of group 12; the transform then happens in fixed function + * after emitvtx (gl-re/hw-tcl-enabling.md). + * + * On by default. It is proven in circuit - scaling the group 8 floats by half + * halves the rendered box, while the same perturbation on the software path + * changes nothing - and it is what frees record dwords 8..13, so a fragment + * program with a varying no longer has to be kept off the part's transform. + * SGX_MTE_VP=0 puts the viewport back in the shader epilogue, which is the + * arrangement every measurement before 2026-08-28 was taken on. + * + * Here rather than in sgx_pipe.c because both layers ask, and the host tests + * build this file without that one. */ +/* Where diagnostics go. The X server closes the driver's stderr, so every + * sgx_dbg() from inside it is lost and the whole 2D path has to be debugged + * blind - which is why SGX_LOG names a file instead. Opened once, appended to, + * line buffered so a lockup still leaves the last line on disk. */ +FILE *sgx_log(void) +{ + static FILE *f; + static int tried; + + if (!tried) { + const char *e = SGX_ENVS("SGX_LOG"); + + tried = 1; + if (e && *e) { + f = fopen(e, "ae"); + if (f) + setvbuf(f, NULL, _IOLBF, 0); + } + } + return f ? f : stderr; +} + +int sgx_mte_viewport(void) +{ + static int v = -1; + + if (v < 0) { + const char *e = SGX_ENVS("SGX_MTE_VP"); + + v = !(e && *e == '0'); + } + return v; +} + +static uint32_t sgx_ctx_draw_cmd(const struct sgx_context *ctx, uint32_t n); +static uint32_t sgx_ctx_draw_cmd_for(const struct sgx_context *ctx, + unsigned objtype, uint32_t n); +static unsigned sgx_ctx_prim_width(const struct sgx_context *ctx); + +static const struct sgx_cookie_caps sgx_caps = { 0, 0 }; + +/* Which window each object belongs in. Not an allocator choice - the hardware + * reaches them through different requestors. */ +static const uint32_t sgx_ctx_window[SGX_CTX_BO_COUNT] = { + [SGX_CTX_BO_PARAM] = SGX_VM_PARAM, + [SGX_CTX_BO_HEAP] = SGX_VM_PDS, + [SGX_CTX_BO_TARGET] = SGX_VM_SURFACE, + [SGX_CTX_BO_RASTGEOM] = SGX_VM_RASTGEOM, + [SGX_CTX_BO_USSE] = SGX_VM_PDS, + [SGX_CTX_BO_VTX] = SGX_VM_SCENE, + [SGX_CTX_BO_DEPTH] = SGX_VM_SURFACE, + [SGX_CTX_BO_TEX] = SGX_VM_SURFACE, + [SGX_CTX_BO_TEX2] = SGX_VM_SURFACE, +}; + +/* Where each object has to sit. The heap and the shader code share a window + * but not an address, and the frame's relocations are generated against these + * exact addresses, so they are fixed rather than allocated. */ +static const uint64_t sgx_ctx_va[SGX_CTX_BO_COUNT] = { + /* Moved with its window, which moved to make room for the raster + * geometry one below it. */ + [SGX_CTX_BO_PARAM] = 0x31000000ull, + [SGX_CTX_BO_HEAP] = 0x20010000ull, + [SGX_CTX_BO_TARGET] = 0x80000000ull, + [SGX_CTX_BO_RASTGEOM] = 0x30000000ull, + [SGX_CTX_BO_USSE] = 0x20800000ull, + [SGX_CTX_BO_VTX] = 0x40000000ull, + [SGX_CTX_BO_DEPTH] = 0x81200000ull, + [SGX_CTX_BO_TEX] = 0x82400000ull, + [SGX_CTX_BO_TEX2] = 0x83a00000ull, +}; + +/* A second address for the colour target, halfway to the depth window's own + * start. A double buffered client alternates two objects, and binding the + * incoming one here leaves the outgoing one exactly where the render still + * running into it expects to find it - which is what lets the frame be built + * without waiting for that render. Nine megabytes each; a target larger than + * that keeps the single slot and the wait. */ +#define SGX_CTX_TARGET_VA2 0x80900000ull +#define SGX_CTX_TARGET_SLOT 0x00900000ull + +/* Two target slots, off by default: it changes when a frame may be built + * against a render still running, which is the kind of thing that renders a + * plausible wrong picture rather than failing. */ +static int sgx_target_slots(void) +{ + static int on = -1; + + if (on < 0) + on = SGX_ENV("SGX_TARGET_SLOTS"); + return on; +} + +/* The four surface slots above are sized to the largest thing the caps allow + * to be bound at each: 18 MiB for the render target and the depth buffer, + * which at 2048x2048 are 17 with the raster pass's slack, and 22 MiB for the + * two texture units, which at 2048x2048 with a mip chain are 21. That is 80 + * MiB of the 128 MiB surface window; the winsys bump-allocates above them, + * from 0x85000000, and the two tables have to agree. + * + * They used to be 8 MiB apart, and that was the real ceiling however large the + * advertised cap was: both a 2048 target and a 2048 chain ran into the next + * slot and the hardware faulted on an address the object no longer owned. + * Making them 24 MiB each went too far the other way and starved the bump + * region, which is where every resource is first bound. Nothing else names + * these addresses - every descriptor is relocated from what the bind + * returned. */ + +/* The sizes the captured frame's buffers had; xpsb_frame_bufs carries the + * same figures for the buffers that are not sized by the framebuffer. */ +static const uint64_t sgx_ctx_size[SGX_CTX_BO_COUNT] = { + [SGX_CTX_BO_HEAP] = 7864320ull, + [SGX_CTX_BO_RASTGEOM] = 98304ull, + /* Each draw record gets its own program slot in here, so this bounds + * how many draws one frame can hold - at 128 KiB it was ninety, and a + * game frame is hundreds. The PDS window runs to 0x21000000, which + * leaves eight megabytes above the buffer's base. */ + [SGX_CTX_BO_USSE] = 2097152ull, + [SGX_CTX_BO_VTX] = 4554752ull, + [SGX_CTX_BO_TEX] = 98304ull, + [SGX_CTX_BO_TEX2] = 98304ull, +}; + +int sgx_context_init(struct sgx_context *ctx, struct sgx_winsys *ws) +{ + if (!ctx || !ws) + return -EINVAL; + memset(ctx, 0, sizeof(*ctx)); + ctx->fb.cbuf_format = SGX_CBUF_FMT_NONE; + ctx->ws = ws; + ctx->isp_dirty = 1; + ctx->colormask = 0xfu; + ctx->vis_reg = -1; /* no query counting */ + { + /* A diagnostic knob: each draw record is a primitive block and + * the DPM allocates parameter memory per block, so a frame + * with several may need a bigger heap than one with one. */ + const char *e = SGX_ENVS("SGX_MAX_PRIMS"); + + ctx->max_prims = (e && *e) ? (uint32_t)strtoul(e, NULL, 0) : + SGX_DPM_DEFAULT_MAX_PRIMS; + if (!ctx->max_prims) + ctx->max_prims = SGX_DPM_DEFAULT_MAX_PRIMS; + } + /* The submit comes back once the render is fired, so the application + * builds its next frame while the core is still on this one. Every + * point that then has to see the result waits for it: the head of the + * next flush, every resource map, and pipe_screen::fence_finish. + * + * SGX_ASYNC=0 puts the wait back inside the ioctl; sgx_async=0 on the + * kernel module does the same for a machine that cannot be rebuilt. */ + { + const char *e = SGX_ENVS("SGX_ASYNC"); + + sgx_winsys_set_async(ws, !e || *e != '0'); + } + ctx->clear_color = XPSB_CLEAR_DEFAULT; + ctx->clear_depth = 0x3f800000u; + ctx->blend_op = SGX_BLEND_NONE; + /* Off until a caller asks. Every frame used to begin by clearing the + * target, which is right for something that redraws the whole picture + * every time and wrong for everything else: a display server draws + * one rectangle and expects the rest of the screen to still be there, + * and instead each frame wiped the one before it. */ + ctx->clear_on = 0; + ctx->zclear_on = 0; + ctx->line_width = 1.0f; + ctx->vtx_floats = SGX_VTX_FLOATS; + /* The render target, the depth buffer and the frame's own texture have + * fixed addresses in the surface window, and the frame's relocations + * name them. Anything the caller allocates - a texture of its own - + * has to come after them, or the first allocation is handed the render + * target's address. */ + sgx_winsys_reserve(ws, SGX_VM_SURFACE, + sgx_ctx_va[SGX_CTX_BO_TEX2] + + sgx_ctx_size[SGX_CTX_BO_TEX2] + 0xfffull); + return 0; +} + +void sgx_context_fini(struct sgx_context *ctx) +{ + unsigned i; + + if (!ctx) + return; + /* The context owns its four objects, so it releases them. Zeroing the + * struct without this leaves them bound in the kernel for the lifetime + * of the file descriptor. */ + for (i = 0; i < SGX_CTX_BO_COUNT; i++) + if (ctx->bound[i]) + sgx_bo_free(ctx->ws, &ctx->bo[i]); + free(ctx->range); + free(ctx->vtx_row); + ctx->vtx_row = NULL; + ctx->vtx_row_cap = 0; + free(ctx->uni_set); + free(ctx->uni_set_dwords); + free(ctx->fs_uni_set); + free(ctx->fs_uni_set_dwords); + memset(ctx, 0, sizeof(*ctx)); +} + +/* Whether a texture keeps an address of its own instead of being cycled + * through the unit's fixed one. Read here as sgx_pipe.c reads it. */ +static int sgx_tex_keep_va(void) +{ + static int v = -1; + + if (v < 0) + v = SGX_ENVS("SGX_TEX_KEEP_VA") != NULL; + return v; +} + +static int ctx_alloc(struct sgx_context *ctx, enum sgx_ctx_bo which, + uint64_t size) +{ + int ret; + + if (ctx->bound[which]) { + if (ctx->bo[which].size >= size) + return 0; + /* The caller's window grew. These live at fixed addresses and + * the raster pass writes the whole target, so keeping the + * smaller buffer means writing past the end of its mapping - + * an MMU fault, which on this part is a lock that needs the + * power button. Replace it at the size now asked for. */ + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: ctx_alloc: slot %d grows from " + "%llu to %llu byte(s)\n", (int)which, + (unsigned long long)ctx->bo[which].size, + (unsigned long long)size); + sgx_bo_free(ctx->ws, &ctx->bo[which]); + ctx->bound[which] = 0; + } + ret = sgx_bo_new(ctx->ws, &ctx->bo[which], size, 0); + if (ret) { + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: ctx_alloc: slot %d, %llu bytes: " + "allocation refused %d\n", (int)which, + (unsigned long long)size, ret); + return ret; + } + ret = sgx_bo_bind_at(ctx->ws, &ctx->bo[which], sgx_ctx_window[which], + sgx_ctx_va[which]); + /* The canonical address first, because a relocation that can be read + * against a known number is worth keeping. But a second context in the + * same file has the same canonical address, so a slot large enough to + * reach its neighbour's comes back EEXIST - and a slot that fails to + * bind is a frame that is never built, which reads as a frame rate + * rather than as an error. Every address here is relocated through the + * frame's buffer table, so anywhere in the same window will do. */ + if (ret) { + int any = sgx_bo_bind(ctx->ws, &ctx->bo[which], + sgx_ctx_window[which]); + + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: ctx_alloc: slot %d, %llu bytes " + "at 0x%llx: bind refused %d, anywhere %d -> " + "0x%llx\n", (int)which, + (unsigned long long)size, + (unsigned long long)sgx_ctx_va[which], ret, any, + (unsigned long long)ctx->bo[which].gpu_va); + if (any) { + /* Loudly, always. A slot that fails to bind is a frame + * that is never built, and this driver then reports a + * frame rate for frames it never submitted - 26.4 fps + * drawing nothing, measured, against 13.6 drawing the + * game. A silent -EEXIST here is worth more confusion + * than every diagnostic in this file. */ + fprintf(sgx_log(), "sgx: slot %d could not be bound at " + "0x%llx (%d) or anywhere in its window (%d) - " + "nothing this context draws will appear\n", + (int)which, + (unsigned long long)sgx_ctx_va[which], ret, + any); + return ret; + } + } + ctx->bound[which] = 1; + return 0; +} + +/* The heap the DPM bins a scene into, sized as sgxtri sizes it. + * + * The kernel owns the heap now - it has to, because the hardware takes one + * heap load per boot and every client must share the memory that load + * describes. This is kept because it is the reasoning about how big the heap + * has to be, and the kernel's fixed size is checked against it. */ +static uint64_t __attribute__((unused)) +sgx_dpm_param_bytes(const struct sgx_context *ctx) +{ + uint64_t work = (uint64_t)ctx->cookie[1] * SGX_DPM_BYTES_PER_MTILE + + (uint64_t)ctx->max_prims * SGX_DPM_BYTES_PER_PRIM * + SGX_DPM_PRIM_MTILE_SPREAD; + uint32_t pages = (uint32_t)((work + 4095u) >> 12); + uint32_t est, cap, first, room; + + if (pages < SGX_DPM_MIN_WORK_PAGES) + pages = SGX_DPM_MIN_WORK_PAGES; + est = pages + SGX_DPM_RESERVE_PAGES; + + /* Map more than the kernel hands the DPM. The tiler overruns the + * estimate it is given, and sgxtri leaves it room to: it maps four + * times as much and reports the estimate alone. Mapping exactly what + * the DPM is told means the first page past the estimate is not + * mapped, so the overrun is an MMU fault a page or two past the end - + * with no out-of-memory event, because from the DPM's side there was + * still room. That is what a heavy GL client hit. + * + * Two things bound the cap: the DPM packs a page index in sixteen + * bits, counted from the base of the 256 MiB region the heap sits in, + * and the window itself. */ + cap = est * SGX_DPM_CAP_MULTIPLIER; + first = (uint32_t)((sgx_ctx_va[SGX_CTX_BO_PARAM] & 0x0fffffffu) >> 12); + room = SGX_DPM_MAX_PAGE_INDEX > first ? + SGX_DPM_MAX_PAGE_INDEX - first : 0u; + if (cap > room) + cap = room; + { + uint32_t wpages = (uint32_t)(sgx_window_bytes( + (uint32_t)sgx_ctx_window[SGX_CTX_BO_PARAM]) >> 12); + + if (wpages && cap > wpages) + cap = wpages; + } + if (cap < SGX_DPM_MIN_CAP_PAGES) { + uint32_t want = SGX_DPM_MIN_CAP_PAGES; + + if (want > room) + want = room; + { + uint32_t wp = (uint32_t)(sgx_window_bytes( + (uint32_t)sgx_ctx_window[SGX_CTX_BO_PARAM]) >> 12); + + if (wp && want > wp) + want = wp; + } + if (cap < want) + cap = want; + } + if (cap < est) + cap = est; + return (uint64_t)cap << 12; +} + +/* The frame's static content: the state heap, the shader code, the raster + * geometry, the draw records and the geometry itself. sgxtri's frame_build() + * writes the same buffers; this is that sequence with the parts driven by + * environment variables and by its own capture hooks left out. */ +/* Dwords the parameter stream carries ahead of the first draw record: the + * input-state record, when the uniforms live in the secondary bank. */ +/* SGX_TA_INBLOCK puts the input-state record where the vendor puts it - + * inside the draw block, between the state pair and the index list - instead + * of at the head of the parameter stream. A valid record naming nothing but a + * HALT still kills every triangle when it sits at the head, so the position is + * what the parser objects to, not the program it names. Then there is no head + * shift: the pair is spliced in after the relocations have run. */ +static int ctx_ta_inblock(void) +{ + static int v = -1; + + if (v < 0) + v = SGX_ENVS("SGX_TA_INBLOCK") != NULL; + return v; +} + +static unsigned ctx_ta_prefix(const struct sgx_context *ctx) +{ + return (ctx->hwtcl && sgx_hwtcl_ta_record() && !ctx_ta_inblock()) ? + SGX_HWTCL_TA_PREFIX_DW : 0u; +} + +/* How large the object behind a texture unit's fixed address has to be. The + * unit's address is fixed and a draw record keeps the descriptor it was built + * with, so the fallback has to cover the largest extent any descriptor has + * named there - otherwise a record outliving its caller's texture samples past + * the smaller object and faults the texture cache, which is what ioquake3 did + * once or twice a minute. */ +static uint64_t ctx_tex_fallback_size(const struct sgx_context *ctx, + unsigned unit) +{ + enum sgx_ctx_bo slot = unit ? SGX_CTX_BO_TEX2 : SGX_CTX_BO_TEX; + uint64_t want = sgx_ctx_size[slot]; + unsigned u; + + /* The largest extent named at any unit, not only at this one. Eight + * units share two fixed addresses, so every unit above the first + * names the second object, and a record outliving its caller's + * texture can name any of their extents at whatever is mapped there. + * Sized from one unit's own need the object was a fifth of what a + * record asked for, and the texture cache faulted at 0x82555000 on + * every glmark2 terrain frame. */ + for (u = 0; u < SGX_MAX_TEX_UNITS; u++) + if (ctx->tex_need[u] > want) + want = (ctx->tex_need[u] + 0xfffull) & ~0xfffull; + return want; +} + +static int ctx_build_frame_inner(struct sgx_context *ctx); + +/* Timed as one section: it runs once per submit, so at ioquake3's six submits + * a swap it runs six times a displayed frame. */ +static int ctx_build_frame(struct sgx_context *ctx) +{ + uint64_t t = sgx_perf_mark(); + int r = ctx_build_frame_inner(ctx); + + sgx_perf_add(0, t); + return r; +} + +static int ctx_build_frame_inner(struct sgx_context *ctx) +{ + struct xpsb_surface_desc dest, tex; + struct xpsb_quad q; + uint32_t w0, w1; + float vtx[4 * XPSB_VTX_STRIDE_MAX]; + uint16_t idx[6]; + uint32_t *heap, *use; + char *v; + int ret; + unsigned i; + + for (i = 0; i < SGX_CTX_BO_COUNT; i++) { + if (!ctx->bound[i] || ctx->bo[i].map) + continue; + ret = sgx_bo_map(ctx->ws, &ctx->bo[i]); + if (ret) + return ret; + } + heap = ctx->bo[SGX_CTX_BO_HEAP].map; + use = ctx->bo[SGX_CTX_BO_USSE].map; + v = ctx->bo[SGX_CTX_BO_VTX].map; + if (!heap || !use || !v) + return -EINVAL; + /* This rebuilds the template the render reads - the state windows and + * the PDS programs in the heap, the clearing program in the USSE + * buffer, the raster geometry - so a render still on the core has to + * have ended. It is one of the three places a frame's memory is + * written outside the flush, and so one of the three that limit how + * far a deferred wait can be carried. + * + * SGX_UNSAFE_NOWAIT skips it, which is not a mode to run anything in - + * it writes the template under a live render. It exists to bound what + * pipelining the frame is worth before the work of double buffering + * the template is undertaken, since that means making XPSB_HEAP_ADDR a + * runtime address rather than the constant it is. */ + { + uint64_t t = sgx_perf_mark(); + + if (!SGX_ENV("SGX_UNSAFE_NOWAIT")) + sgx_wait_idle(ctx->ws); + sgx_perf_add(1, t); + } + + /* Whether the frame carries a depth test at all follows the caller's + * framebuffer. It was hard-coded on, and has_zsbuf was computed in + * sgx_pipe.c and read nowhere - so a target with no depth attachment + * still got the captured frame's depth state. The per-draw ISP word + * from the depth-stencil state refines it from there. */ + /* SGX_FORCE_DEPTH builds the frame with the depth test even when the + * caller bound no depth attachment, to tell a frame that will not + * rasterise without one apart from a geometry problem. */ + xpsb_gen_heap(heap, (int)ctx->fb.width, (int)ctx->fb.height, + ctx->stride, + (ctx->fb.has_zsbuf || SGX_ENV("SGX_FORCE_DEPTH")) ? 1 : 0); + if (SGX_ENV("SGX_DUMP_BG")) { + unsigned k; + + fprintf(sgx_log(), "sgx: heap 0x0c0 (clear object):"); + for (k = 0x30; k < 0x40; k++) + fprintf(sgx_log(), " %08x", heap[k]); + fprintf(sgx_log(), "\nsgx: heap 0x120 (load-back object):"); + for (k = 0x48; k < 0x58; k++) + fprintf(sgx_log(), " %08x", heap[k]); + fprintf(sgx_log(), "\nsgx: rt descriptor heap[1] %08x heap[3] %08x\n", + heap[1], heap[3]); + } + xpsb_gen_usse(use, ctx->clear_color); + + /* The wide record first, then the program that reads it: installing + * the uniform DMA before the record is described would have it patch + * a vertex PDS program that set_ntex then rewrites. */ + if (ctx->hwtcl) { + xpsb_heap_set_ntex(heap, (unsigned)sgx_hwtcl_ntex()); + ret = sgx_hwtcl_install(heap, use, + sgx_ctx_va[SGX_CTX_BO_HEAP], + SGX_HWTCL_MAT_N); + if (ret) + return ret; + } + { + /* The records are always all there: the user draw needs the + * state the earlier blocks carry, and a stream without them + * leaves the tiler unfinished. What changes is whether the + * clearing ones cover anything. */ + unsigned first = 0u; + + xpsb_gen_rastgeom_from(ctx->bo[SGX_CTX_BO_RASTGEOM].map, + (int)ctx->fb.width, (int)ctx->fb.height, + first); + xpsb_rastgeom_set_z(ctx->bo[SGX_CTX_BO_RASTGEOM].map, first, + ctx->clear_depth); + /* Record 1 is the background object for a frame that does not + * clear, so its geometry has to stay. Blanking it was what + * left the load-back with nothing to paint. */ + unsigned pre = ctx_ta_prefix(ctx); + + /* The same shape set_ntex just gave the fetch: the record's + * granule count and the DMA have to describe one width. */ + ctx->ndraw = xpsb_gen_draw_records_n((uint32_t *)v + pre, + first, + ctx->hwtcl ? + (unsigned)sgx_hwtcl_ntex() : 1u); + ctx->draw_cmd_off = xpsb_draw_cmd_off(first) + pre * 4; + } + + memset(&dest, 0, sizeof dest); + dest.w = ctx->fb.width; + dest.h = ctx->fb.height; + /* Filled in below, once the format says how wide a pixel is. */ + dest.stride = 0; + /* What the caller's colour buffer actually is, when it said. The + * constant here ignored cbuf_format entirely, so the pixel back end + * was configured for the captured frame's format whatever was bound. + * XPSB_FMT_8888 and XPSB_FMT_BGR8888 are the same numbering as + * SGX_FMT_A8R8G8B8 and SGX_FMT_A8B8G8R8, so it passes straight + * through. */ + dest.format = ctx->fb.cbuf_format != SGX_CBUF_FMT_NONE ? + ctx->fb.cbuf_format : XPSB_FMT_8888; + { + unsigned bpp = xpsb_format_bpp(dest.format); + + if (!bpp) + return -EINVAL; + /* At the target's own depth. The stride is kept in pixels and + * was turned into bytes as though every target were four of + * them, so an eight-bit one - a glyph mask - was described + * four times too wide. */ + dest.stride = ctx->stride * bpp; + } + if (xpsb_heap_set_dest(heap, &dest)) + return -EINVAL; + + /* The frame issues a texture fetch whatever the program does - ntex is + * never zero and the relocation path refuses less - so the unit's + * address must have something behind it or the render faults. The + * context's own object is that backing and nothing more. + * + * It used to carry a blue and white checkerboard copied from the + * bare-metal cube's own texture, which turned "no texture bound" into + * plausible output rather than a failure: a moved window rendered as + * that checkerboard, and it took tracing the pattern back to its own + * formula to see it was ours. */ + memset(&tex, 0, sizeof tex); + tex.w = 64; + tex.h = 64; + tex.stride = 64 * 4; + tex.format = XPSB_FMT_8888; + tex.umode = XPSB_WRAP_REPEAT; + tex.vmode = XPSB_WRAP_REPEAT; + tex.minfilter = XPSB_FILTER_NEAREST; + tex.magfilter = XPSB_FILTER_NEAREST; + if (xpsb_tex_state_mode(&w0, &w1, &tex, XPSB_TEXCTL_CAPTURE)) + return -EINVAL; + heap[XPSB_TEXCTL_OFF / 4] = w0; + heap[XPSB_TEXSTATE_OFF / 4] = w1; + + /* One triangle from the quad generator's four vertices, the index + * count cut to three. Until set_vertex_buffers is wired this is the + * geometry every draw gets. + * + * It covers nothing, because it was being rasterised. The capture's + * eight-pixel inset makes a triangle that on a 64x64 surface is + * (8,8)-(56,8)-(8,56), and that is exactly the 1128 pixels in the box + * 8,8..54,54 a clear used to leave black - measured at four insets, + * 2016 / 1128 / 496 / 120 pixels, each the triangle's own area, and 0 + * once it is collapsed. Nothing else needs it: every real draw brings + * its own geometry, and the four load-back cases pass with the whole + * suite otherwise unchanged. SGX_TMPL_INSET restores it. */ + memset(&q, 0, sizeof q); + { + const char *e = SGX_ENVS("SGX_TMPL_INSET"); + + if (e) { + float in = (float)strtod(e, NULL); + + q.x0 = in; + q.y0 = in; + q.x1 = (float)ctx->fb.width - in; + q.y1 = (float)ctx->fb.height - in; + } else { + q.x0 = q.x1 = q.y0 = q.y1 = 0.0f; + } + } + q.u1 = 1.0f; q.v1 = 1.0f; + q.r = 1.0f; q.g = 1.0f; q.b = 1.0f; q.a = 1.0f; + /* One width for both: generated at eleven floats and copied at + * fourteen, the wide record's vertices were laid out at the wrong + * stride and its tail came from uninitialised stack. */ + { + unsigned ntex = ctx->hwtcl ? 2u : 1u; + + unsigned gs = xpsb_vtx_stride(ntex); + unsigned rs = ctx->vtx_floats ? ctx->vtx_floats : gs; + unsigned kv, ncopy = gs < rs ? gs : rs; + + memset(vtx, 0, sizeof vtx); + if (!xpsb_gen_quad_n(vtx, idx, &q, ntex, NULL)) + return -EINVAL; + /* At the width the frame declares, not the generator's. The + * quad comes out on xpsb_vtx_stride()'s three-float sets - + * fourteen dwords - while a frame carrying three four-float + * sets reads records twenty apart, so its corners were fetched + * from the wrong offsets and one arrived with w of zero, which + * the rasteriser runs to the corner of the viewport. Only a + * frame with no record of its own draws this triangle, and a + * clear with no draw behind it opens exactly one: that is the + * needle glmark2's refract drew from its bunny on every frame + * after the first. */ + memset(v + XPSB_VTX_OFF, 0, 4u * rs * sizeof vtx[0]); + for (kv = 0; kv < 4u; kv++) + memcpy((float *)(v + XPSB_VTX_OFF) + kv * rs, + vtx + kv * gs, ncopy * sizeof vtx[0]); + } + memcpy(v + XPSB_IDX_OFF, idx, sizeof idx); + ctx->nidx = 3; + *(uint32_t *)(v + ctx->draw_cmd_off) = + sgx_ctx_draw_cmd(ctx, ctx->nidx); + return 0; +} + +int sgx_set_framebuffer(struct sgx_context *ctx, + const struct sgx_framebuffer *fb) +{ + uint64_t target_size, depth_size; + uint32_t samples, axis; + int ret; + + if (!ctx || !fb || !fb->width || !fb->height) + return -EINVAL; + /* EURASIA_RENDERSIZE_MAXX/MAXY. This said 4096, which is what the + * scene cookie's twelve-bit extent fields hold, not what the ISP + * renders - and a claim the part cannot honour is worse than a + * refusal, because the caller gets a target it believes in. */ + if (fb->width > XPSB_RENDER_SIZE_MAX || + fb->height > XPSB_RENDER_SIZE_MAX) + return -EINVAL; + /* One sample or 2x2, and refused rather than rounded: everything from + * the region-header array to the ZLS extent is sized from this, and a + * count the scene was not built for has the tiler run past the end of + * the array. */ + samples = fb->samples ? fb->samples : 1u; + axis = xpsb_msaa_axis(samples); + if (!axis) + return -EINVAL; + /* At 2x2 the ISP's render box is in sample tiles and its field is + * eight bits a side, so the largest square a multisampled frame + * reaches is short of the 2048 a single-sampled one does. Checked + * here so the refusal names the framebuffer rather than surfacing as + * a frame that builds and then fires at a truncated box. */ + if (axis > 1 && + (((fb->width + 15u) / 16u) * axis > 0xffu || + ((fb->height + 15u) / 16u) * axis > 0xffu)) + return -EINVAL; + + + /* Only when the target really changes. Depth stored at another + * target's extent and stride would be read back at the wrong layout, + * so it has to be dropped then - but Gallium re-sets the same + * framebuffer routinely, and dropping it every time meant a pass that + * continues a picture never loaded depth back. The feature suite's + * depth test saw it as soon as frames began splitting on a program + * change: the second pass started every tile at the background depth + * and the far quad drew over the near one. */ + { + int same = ctx->cookie_valid && + ctx->fb.width == fb->width && + ctx->fb.height == fb->height && + ctx->fb.cbuf_format == fb->cbuf_format && + ctx->fb.has_zsbuf == fb->has_zsbuf && + ctx->fb.samples == samples; + + if (!same) + ctx->depth_stored = 0; + } + ctx->fb = *fb; + ctx->fb.samples = samples; + /* The scene at the sample resolution: the tail-pointer and + * region-header arrays hold one entry per sample tile, so both grow + * by four at 2x2. The kernel builds its own copy of this from the + * submit's SGX_SUBMIT_MSAA_2X2 flag; this one sizes the heap the + * context asks for. */ + if (sgx_scene_info_ms(&sgx_caps, fb->width, fb->height, axis, axis, + ctx->cookie, &ctx->scene_size, &ctx->clear_start, + &ctx->clear_pages)) + return -EINVAL; + ctx->cookie_valid = 1; + /* And the kernel, which sizes its scene from the submit's flag rather + * than from anything in the streams. */ + if (ctx->ws && sgx_winsys_set_msaa(ctx->ws, samples)) + return -EINVAL; + + /* How many triangles this target may bin. The kernel hands the DPM a + * quarter of the mapped heap - the ratio the heap is allocated at, so + * the tiler has room to overrun - and holds SGX_DPM_RESERVE_PAGES of + * that back; what is left, less the macrotile working set, is + * parameter memory at SGX_DPM_BYTES_PER_PRIM per primitive per + * macrotile it spreads across. + * + * This was a flat four thousand and ninety-six, which is a quarter of + * what the heap actually holds: glmark2's models are seven thousand + * triangles and every scene using one was refused outright, so the + * frame was dropped and the clear was all that reached the screen. */ + /* The heap is mapped SGX_DPM_CAP_MULTIPLIER times the size the DPM is + * told, and that headroom is exactly the overrun the macrotile spread + * describes - so the spread is not multiplied in here. A flat 4096 was + * a quarter of what the heap holds and refused a seven-thousand + * triangle model outright. + * + * 6d90f50 divided by the spread as well, on the reading that this was + * what drew glmark2 refract's needle. It was not - bb7ce71 found the + * needle in the frame's built-in triangle - and the division cost + * build 17 fps to 12 and shading 11 to 7 for nothing, measured on one + * binary from a clean boot with SGX_MAX_PRIMS either way. */ + if (!SGX_ENV("SGX_MAX_PRIMS")) { + uint64_t avail = (uint64_t)(SGX_PARAM_HEAP_PAGES / + SGX_DPM_CAP_MULTIPLIER); + + if (avail > SGX_DPM_RESERVE_PAGES) { + uint64_t bytes = (avail - SGX_DPM_RESERVE_PAGES) << 12; + uint64_t mt = (uint64_t)ctx->cookie[1] * + SGX_DPM_BYTES_PER_MTILE; + + bytes = bytes > mt ? bytes - mt : 0; + bytes /= SGX_DPM_BYTES_PER_PRIM; + if (bytes > SGX_DPM_DEFAULT_MAX_PRIMS) + ctx->max_prims = (uint32_t)bytes; + } + } + + /* The parameter heap is what the DPM bins into, and it is not the + * scene: sizing it to scene_size gave two pages, which + * sgx_ta_mem_info() refuses outright. */ + /* The parameter heap is the kernel's, not this context's. The DPM takes + * one heap load per boot, so the heap it describes has to be the same + * memory for every client - a per-context heap gave the second client + * different pages behind the same address and a page table the load + * never filled. The kernel binds its own into this address space. */ + ret = ctx_alloc(ctx, SGX_CTX_BO_HEAP, sgx_ctx_size[SGX_CTX_BO_HEAP]); + if (ret) + return ret; + ret = ctx_alloc(ctx, SGX_CTX_BO_RASTGEOM, + sgx_ctx_size[SGX_CTX_BO_RASTGEOM]); + if (ret) + return ret; + ret = ctx_alloc(ctx, SGX_CTX_BO_USSE, sgx_ctx_size[SGX_CTX_BO_USSE]); + if (ret) + return ret; + ret = ctx_alloc(ctx, SGX_CTX_BO_VTX, sgx_ctx_size[SGX_CTX_BO_VTX]); + if (ret) + return ret; + /* Also when what is already there is too small for the largest extent + * a record can name at the address - ctx_alloc() replaces it. Only + * while the context's own object is the one bound: a caller's texture + * holds the address itself, and allocating over that is refused. */ + if (!ctx->tex_bo[0] || + ctx->bo[SGX_CTX_BO_TEX].size < ctx_tex_fallback_size(ctx, 0)) { + ret = ctx_alloc(ctx, SGX_CTX_BO_TEX, + ctx_tex_fallback_size(ctx, 0)); + if (ret && !ctx->tex_bo[0]) + return ret; + } + /* The second unit's descriptor is read whether or not anything samples + * it, so it needs something mapped behind it - and two units no longer + * imply hardware transform now that they can share one coordinate + * set, so it is allocated either way. Unless a caller's texture is + * already bound there: the unit's address is fixed, so allocating over + * it was refused with EEXIST, the framebuffer went with it, and every + * draw that bound a second texture was dropped. That is every masked + * composite the X server makes - each glyph and each list row. */ + /* Skipping this whenever any texture is on unit 1 is right only while + * that texture holds the unit's fixed address. With SGX_TEX_KEEP_VA a + * texture keeps an address of its own, so nothing is mapped there, the + * frame's unit-1 descriptor resolves to zero and a record carrying no + * second texture reads address zero - the texture-cache fault that + * ends a level after about a minute. Only in that mode: on the fixed + * address path the caller's texture is the unit's address and + * allocating over it is refused. */ + if (!ctx->tex_bo[1] || + ctx->bo[SGX_CTX_BO_TEX2].size < ctx_tex_fallback_size(ctx, 1) || + (sgx_tex_keep_va() && + ctx->tex_bo[1]->gpu_va != sgx_ctx_va[SGX_CTX_BO_TEX2])) { + ret = ctx_alloc(ctx, SGX_CTX_BO_TEX2, + ctx_tex_fallback_size(ctx, 1)); + if (ret && !ctx->tex_bo[1]) + return ret; + } + + /* The raster pass writes a surface wider than the framebuffer, so the + * target and the depth buffer carry the same padding sgxtri gives + * them: stride rounded to 32, and 32 rows beyond the height. */ + ctx->stride = ctx->target_is_scanout ? ctx->scanout_pitch / 4 + : ((ctx->fb.width + 31u) & ~31u); + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: stride %u for %ux%u (scanout %d, " + "scanout_pitch %u)\n", ctx->stride, ctx->fb.width, + ctx->fb.height, ctx->target_is_scanout, + ctx->scanout_pitch); + target_size = (uint64_t)ctx->stride * (ctx->fb.height + 32u) * 4; + /* Depth holds a value per sample and colour does not: the pixel back + * end downscales colour on store, so the target stays the size the + * caller asked for while the depth surface grows by the sample + * count. */ + depth_size = (uint64_t)ctx->stride * axis * + ((uint64_t)ctx->fb.height + 32u) * axis * 4; + /* Not over the caller's own depth attachment, which holds the slot + * the way a caller's colour buffer holds the target's. */ + if (!ctx->depth_bo) { + ret = ctx_alloc(ctx, SGX_CTX_BO_DEPTH, depth_size); + if (ret) + return ret; + } else if (ctx->bo[SGX_CTX_BO_DEPTH].size < depth_size) { + /* It was big enough for the framebuffer it was attached to + * and is not for this one; the frame would write past it. */ + sgx_release_depth(ctx); + ret = ctx_alloc(ctx, SGX_CTX_BO_DEPTH, depth_size); + if (ret) + return ret; + } + if (!ctx->target_is_scanout && !ctx->target_bo) { + ret = ctx_alloc(ctx, SGX_CTX_BO_TARGET, target_size); + if (ret) + return ret; + } + return ctx_build_frame(ctx); +} + +static void ctx_update_state(struct sgx_context *ctx); + +void sgx_bind_blend(struct sgx_context *ctx, const struct sgx_blend_state *b) +{ + if (!ctx) + return; + ctx->blend_op = b ? b->op : SGX_BLEND_NONE; + /* The instruction the factors compiled to, when they could be. It is + * what the program's final instruction becomes; the named operator is + * only the fallback for what the encoder cannot express. */ + if (b && b->have_insn) + ctx->blend_desc = b->desc; + else + memset(&ctx->blend_desc, 0, sizeof ctx->blend_desc); + ctx->blend_insn = b ? b->have_insn : 0; + /* SRC is the identity - source straight through - so it is no blend at + * all whichever way it was expressed. */ + /* A built instruction is enough on its own: it may be a write mask + * rather than a blend, and a masked draw still has to read the + * destination back for the channels it does not write. */ + /* A logic op is not a blend equation but it wants the same thing from + * the frame: the destination read back, so the program's last + * instruction can have it as an operand. COPY is the identity and + * needs none of that. */ + ctx->logicop = b ? b->logicop : 0; + ctx->logicop_func = b ? b->logicop_func : 0; + if (ctx->logicop && ctx->logicop_func == SGX_LOGICOP_COPY) + ctx->logicop = 0; + ctx->blend_on = ctx->blend_insn || ctx->logicop || + (ctx->blend_op != SGX_BLEND_NONE && + ctx->blend_op != SGX_BLEND_SRC); + /* Blending and translucency are different questions - see + * sgx_blend_translucent(). A named operator the factors could not be + * read from is taken as translucent, which is the safe way round. */ + ctx->blend_translucent = b ? (b->have_insn ? b->translucent : + ctx->blend_on) : 0; + /* The write mask is ISP word B's pass key, so it is state-word input + * like the depth test. */ + ctx->colormask = b ? b->colormask & 0xfu : 0xfu; + ctx->isp_dirty = 1; + ctx_update_state(ctx); +} + +void sgx_set_blend_color(struct sgx_context *ctx, uint32_t packed) +{ + if (ctx) + ctx->blend_const = packed; +} + +void sgx_bind_dsa(struct sgx_context *ctx, const struct sgx_dsa_state *dsa) +{ + if (!ctx || !dsa) + return; + ctx->dsa = *dsa; + ctx->isp_dirty = 1; + ctx_update_state(ctx); +} + +void sgx_bind_rasterizer(struct sgx_context *ctx, + const struct sgx_rasterizer_state *r) +{ + if (!ctx || !r) + return; + ctx->rast = *r; + ctx->isp_dirty = 1; + ctx_update_state(ctx); +} + +/* Recompute the words the state objects produce. Gallium binds state far less + * often than it draws, so this is where the translation belongs rather than in + * the draw path. */ +static void ctx_update_state(struct sgx_context *ctx) +{ + if (!ctx->isp_dirty) + return; + sgx_isp_sets(&ctx->dsa, &ctx->rast, ctx->hwtcl, ctx->colormask, + &ctx->isp_ff, &ctx->isp_bf); + ctx->raster_word = sgx_raster_word(&ctx->rast); + ctx->isp_dirty = 0; +} + +/* The task's temporary-register count, low five bits. The rest is in + * sgx_temp_field_hi(), so 31 is not the ceiling. + * The unit is one register per count, not a quad of them: the closed driver's + * packed-YUV blit uses r0..r5 and writes 6 (XPSB_VID_TEMPS), its composite + * programs use none and write 0, and its hardware-transform vertex program + * writes 28 - every one of them the register count, none of them a quarter of + * it. SGX_TEMP_UNIT=quad exists only to re-run that comparison; it would + * under-allocate fourfold and is contradicted by the evidence above. + * + * What a fragment program pays before its own values is the scratch it + * actually reaches plus the output stand-in, in alloc_regs(); that used to be + * a flat twenty registers, which is where the headroom for the X server's + * shaders came from. */ +unsigned sgx_temp_field(unsigned ntemps) +{ + return ntemps & 0x1fu; +} + +/* The high part. The closed driver writes the count in two places - the low + * five bits as count << 27 in ds0[1], and 0x20 | (count >> 5) in ds0[8] - and + * the captured frame agrees: its ds0[8] is 0x20, which is that expression for + * a program using none. Writing only the low bits capped us at thirty-one and + * refused the X server's glyph composite; the closed driver's own colour count + * is ninety-six. */ +unsigned sgx_temp_field_hi(unsigned ntemps) +{ + return 0x20u | (ntemps >> 5); +} + +/* How many temporaries a task may be told about: the closed driver's own + * budget of ninety-six, programmed across two PDS data dwords as reversed in + * gl-re/usse-register-pressure/. + * + * The feature suite's hitemp case settles that the pair is honoured. It needs + * exactly 32, so the low five bits are zero and the whole count rides in the + * high word, and it renders its expected colour to the byte - which it could + * not do if that word were ignored. Capping at 31 refused such a program + * instead, and that is what took X down: glamor's composite needs 40, its link + * failed, and the block handler then ran against a torn-down GL context. */ +/* SGX_SA_MAX raises the bank size the driver believes in, so a program that + * overflows it can be run anyway - which says whether fitting a shader into + * the bank would help a caller, before the work of making it fit. */ +unsigned sgx_sa_max(void) +{ + static int n = -1; + + if (n < 0) { + const char *e = SGX_ENVS("SGX_SA_MAX"); + + n = (e && *e) ? atoi(e) : (int)SGX_SA_MAX; + if (n < 4) + n = (int)SGX_SA_MAX; + } + return (unsigned)n; +} + +unsigned sgx_temp_max(void) +{ + return SGX_TEMP_FIELD_MAX; +} + +int sgx_temp_ok(unsigned ntemps) +{ + return ntemps <= sgx_temp_max(); +} + +/* Where a fragment program too long for the captured slot goes: above the + * captured USSE image, which ends near 0x2060, and below the per-record group + * region at 0x8000. Sixteen kilobytes is two thousand instructions. */ +#define SGX_USSE_FRAG_LONG 0x4000u +#define SGX_USSE_FRAG_LONG_END 0x8000u + +static int ctx_use_word(struct sgx_context *ctx, uint32_t usse_off, + uint32_t *out); +static void ctx_set_mte_viewport(struct sgx_context *ctx); + +/* What the bound programs need in the sa bank. Each program's nsecattr is its + * own requirement counted from sa0, so the bank has to satisfy the larger of + * the two rather than their sum - they are not concatenated, they overlap. */ +static void ctx_update_sa(struct sgx_context *ctx) +{ + unsigned v = ctx->vs ? ctx->vs->nsecattr : 0; + unsigned f = ctx->fs ? ctx->fs->nsecattr : 0; + + ctx->sa_dwords = v > f ? v : f; +} + +/* The colour buffer as a binary PPM, so a frame can be looked at rather than + * sampled. The target is A8R8G8B8 and the file is RGB. */ +void sgx_dump_target(struct sgx_context *ctx, const char *path) +{ + const uint32_t *px; + unsigned w, h, y, x; + FILE *f; + + if (!ctx || !path) + return; + /* The pixels are only there once the render has ended, and with a + * deferred wait the flush that produced them has already returned. */ + sgx_wait_idle(ctx->ws); + /* The colour buffer is not mapped for the CPU in normal use. */ + if (!ctx->bo[SGX_CTX_BO_TARGET].map && + sgx_bo_map(ctx->ws, &ctx->bo[SGX_CTX_BO_TARGET])) { + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: dump: the target would not map\n"); + return; + } + w = ctx->fb.width; + h = ctx->fb.height; + if (!w || !h) + return; + { + /* One file per frame: a scene's frames differ, and comparing + * the wrong one against what a test sampled is its own kind + * of wrong answer. */ + static unsigned seq; + char name[512]; + + snprintf(name, sizeof name, "%s.%03u", path, seq++); + f = fopen(name, "wb"); + } + if (!f) + return; + px = ctx->bo[SGX_CTX_BO_TARGET].map; + fprintf(f, "P6\n%u %u\n255\n", w, h); + for (y = 0; y < h; y++) + for (x = 0; x < w; x++) { + /* The target's own stride. scanout_pitch is set once + * when a scanout is taken and never cleared, so under + * X every dump of an off-screen pixmap walked rows at + * the screen's pitch and read past the object - which + * made the dump of a 86 wide pixmap pure noise after + * about row twenty. */ + uint32_t v = px[y * (ctx->stride ? ctx->stride : w) + x]; + unsigned char rgb[3]; + + rgb[0] = (unsigned char)(v >> 16); + rgb[1] = (unsigned char)(v >> 8); + rgb[2] = (unsigned char)v; + fwrite(rgb, 1, 3, f); + } + fclose(f); +} + +unsigned sgx_fs_tex_inputs(const struct sgx_context *ctx) +{ + return (ctx && ctx->fs) ? ctx->fs->tex_inputs : 0; +} + +/* The bound fragment program's inputs, by the semantic each one carries. The + * record is filled from the vertex program's outputs and this is what says + * which output belongs to which input. */ +unsigned sgx_fs_inputs(const struct sgx_context *ctx, + const unsigned char **sem, const unsigned char **idx) +{ + if (!ctx || !ctx->fs) + return 0; + if (sem) + *sem = ctx->fs->in_sem; + if (idx) + *idx = ctx->fs->in_sem_idx; + return ctx->fs->nin; +} + +/* What the bound fragment program needs iterated, or NULL when it reads the + * packed colour and nothing else. */ +static void sgx_ctx_dbg(const char *fmt, ...) +{ + va_list ap; + + if (!SGX_ENV("SGX_DEBUG")) + return; + va_start(ap, fmt); + fprintf(sgx_log(), "sgx: "); + vfprintf(sgx_log(), fmt, ap); + va_end(ap); +} + +/* The bound fragment program, so a caller can tell whether the one it laid + * the record out for is the one the draw ran. */ +const struct sgx_shader *sgx_ctx_fs(const struct sgx_context *ctx) +{ + return ctx ? ctx->fs : NULL; +} + +const struct xpsb_attribs *sgx_fs_attribs(const struct sgx_context *ctx) +{ + if (!ctx || !ctx->fs || !ctx->fs->attribs.nset) + return NULL; + return &ctx->fs->attribs; +} + +/* The split the bound fragment program was compiled with, so the record can + * carry both sets. Returns the input that was split, or -1. */ +/* Which fragment input feeds a coordinate set. The sets are numbered with the + * sampled ones first, so this is not the identity, and a record laid out by + * set position fed the wrong varying to the wrong set. */ +/* Which two components of that set's varying carry its coordinate. */ +int sgx_fs_set_coord(const struct sgx_context *ctx, unsigned set, + unsigned char *c0, unsigned char *c1) +{ + if (!ctx || !ctx->fs || set >= ctx->fs->attribs.nset || + set >= XPSB_NSET_MAX) + return -1; + *c0 = ctx->fs->set_c0[set]; + *c1 = ctx->fs->set_c1[set]; + return 0; +} + +int sgx_fs_set_varying(const struct sgx_context *ctx, unsigned set) +{ + if (!ctx || !ctx->fs || set >= ctx->fs->attribs.nset || + set >= XPSB_NSET_MAX) + return -1; + return (int)ctx->fs->set_varying[set]; +} + +/* Which fragment input is handed over as the packed colour, so the record can + * carry that one in the colour slot. */ +/* Sampled sets are numbered first and each takes as many units as sample + * it, so a set's first unit is the sampled count of the sets before it. */ +unsigned sgx_fs_set_depth(const struct sgx_context *ctx, unsigned set) +{ + unsigned i, u = 0; + + if (!ctx || !ctx->fs || set >= ctx->fs->attribs.nset || + set >= XPSB_NSET_MAX) + return 0; + for (i = 0; i < set; i++) + u += ctx->fs->attribs.set[i].sampled; + if (u >= SGX_MAX_SAMPLERS || !(ctx->view_bound & (1u << u))) + return 0; + return ctx->views[u].depth > 1 ? ctx->views[u].depth : 0u; +} + +int sgx_fs_packed_in(const struct sgx_context *ctx) +{ + return (ctx && ctx->fs) ? ctx->fs->packed_in : -1; +} + +int sgx_fs_sets_swapped(const struct sgx_context *ctx) +{ + return (ctx && ctx->fs) ? ctx->fs->sets_swapped : 0; +} + +int sgx_fs_coord_split(const struct sgx_context *ctx, int *set, + unsigned char *c) +{ + if (!ctx || !ctx->fs || ctx->fs->coord_split_in < 0) + return -1; + if (set) + *set = ctx->fs->coord_split_set; + if (c) { + c[0] = ctx->fs->coord_c[0]; + c[1] = ctx->fs->coord_c[1]; + } + return ctx->fs->coord_split_in; +} + +/* Which coordinate set carries gl_FragCoord, or -1 when the program does not + * read it. */ +int sgx_fs_fragcoord_set(const struct sgx_context *ctx) +{ + return (ctx && ctx->fs) ? ctx->fs->fragcoord_set : -1; +} + +/* Floats the record must carry for the bound program's first iterated + * coordinate set, or zero when it reads the packed colour instead. */ +/* The record the bound programs need. A program handed a varying as floats + * carries it at its own width, so the record is wider than the captured + * eleven - and the element list is checked against this before the frame is + * built, so it has to follow the binding rather than the flush. */ +/* The dword a point size adds to the record, if one is riding in it. The + * width the frame declares has to include it or the fetch walks the buffer at + * the wrong stride - measured: the vertices went over 48 bytes wide while the + * frame said 44, and the size the point needed was never read. */ +unsigned sgx_record_extra_dwords(const struct sgx_context *ctx) +{ + return ctx && ctx->point_size_in_record ? 1u : 0u; +} + +/* Every point object this part has is a "UV" form, so the ISP produces a + * coordinate whether or not the program reads one, and ctx_emit_samplers() + * declares a set for it ahead of the program's. Asked in one place because + * three of them have to give the same answer: the frame that declares the + * set, the record's width, and the layout the vbuf emits. */ +int sgx_sprite_uv_set(const struct sgx_context *ctx) +{ + return ctx && ctx->prim_objtype != SGX_ISP_OBJ_TRI && + ctx->prim_objtype != SGX_ISP_OBJ_LINE && + !SGX_ENVS("SGX_NO_SPRITE_UV"); +} + +/* The sets a point object's frame declares: the ISP's own UV ahead of the + * program's own. One function because the frame and the record's width have + * to describe the same list - computed apart, they differed by this set and + * the frame forced the wrong width onto the records. */ +static const struct xpsb_attribs * +ctx_sprite_attribs(const struct xpsb_attribs *a, struct xpsb_attribs *spr) +{ + unsigned q, n = a ? a->nset : 0u; + + memset(spr, 0, sizeof *spr); + if (n + 1u > XPSB_NCOORD_MAX) + n = XPSB_NCOORD_MAX - 1u; + spr->set[0].width = 2; + /* Sampled, not iterated: a texture unit reads it, which is the issue + * the vendor points at TC0. Declared as an iterated varying instead - + * the program reading it as floats - the point rasterises nothing at + * all. That one bit is the difference between a native point and no + * native point. */ + spr->set[0].sampled = 1; + spr->set[0].iterated = 0; + for (q = 0; q < n; q++) + spr->set[q + 1u] = a->set[q]; + spr->nset = n + 1u; + return spr; +} + +void sgx_update_vtx_floats(struct sgx_context *ctx) +{ + if (!ctx || ctx->hwtcl) + return; + /* The sprite's set counts, or this width and the one + * ctx_emit_samplers() computes from the same sets differ by it - and + * the frame then forces the wrong one of the two onto the records. + * That is what kept a sized native point behind SGX_NO_FRAME_STRIDE. */ + if (sgx_sprite_uv_set(ctx)) { + struct xpsb_attribs spr; + + ctx->vtx_floats = xpsb_attrib_stride( + ctx_sprite_attribs(ctx->fs ? &ctx->fs->attribs : NULL, + &spr)) + + sgx_record_extra_dwords(ctx); + return; + } + ctx->vtx_floats = ((ctx->fs && ctx->fs->attribs.nset) ? + xpsb_attrib_stride(&ctx->fs->attribs) : + SGX_VTX_FLOATS) + sgx_record_extra_dwords(ctx); +} + +unsigned sgx_fs_vary_floats(const struct sgx_context *ctx) +{ + if (!ctx || !ctx->fs || !ctx->fs->attribs.nset) + return 0; + if (!ctx->fs->attribs.set[0].iterated || ctx->fs->attribs.set[0].sampled) + return 0; + return ctx->fs->attribs.set[0].width; +} + +const struct sgx_shader *sgx_passthrough_vs(void) +{ + /* Nothing of it is uploaded - the frame carries the program already - + * so it is the empty shader of the right stage, which is what says + * "the frame's own". */ + static const struct sgx_shader vs = { + .stage = UIR_STAGE_VERTEX, + .compiled = 1, + }; + + return &vs; +} + +int sgx_bind_vs(struct sgx_context *ctx, const struct sgx_shader *vs) +{ + if (!ctx) + return -EINVAL; + if (vs && !vs->compiled) + return -EINVAL; + if (vs && vs->stage != UIR_STAGE_VERTEX) + return -EINVAL; + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: bind vs %p nlimm %u ninsns %u\n", + (const void *)vs, vs ? vs->nlimm : 0u, + vs ? vs->ninsns : 0u); + ctx->vs = vs; + ctx_update_sa(ctx); + return 0; +} + +static void ctx_refresh_sampler_state(struct sgx_context *ctx); + +int sgx_bind_fs(struct sgx_context *ctx, const struct sgx_shader *fs) +{ + if (!ctx) + return -EINVAL; + if (fs && !fs->compiled) + return -EINVAL; + if (fs && fs->stage != UIR_STAGE_FRAGMENT) + return -EINVAL; + ctx->fs = fs; + ctx_update_sa(ctx); + sgx_update_vtx_floats(ctx); + ctx_refresh_sampler_state(ctx); + return 0; +} + +int sgx_set_sampler_view(struct sgx_context *ctx, unsigned unit, + const struct sgx_sampler_view *v) +{ + enum sgx_sampler_status st; + + if (!ctx || unit >= SGX_MAX_SAMPLERS) + return -EINVAL; + if (!v) { + ctx->view_bound &= ~(1u << unit); + memset(&ctx->views[unit], 0, sizeof(ctx->views[unit])); + } else { + st = sgx_sampler_view_check(v); + if (st != SGX_SAMPLER_OK) + return -EINVAL; + ctx->views[unit] = *v; + ctx->view_bound |= 1u << unit; + /* The two units have a fixed address, and a draw record keeps + * the descriptor it was built with. So a record can still name + * a caller texture's extent at that address after the + * caller's object has gone and the context's own is mapped + * there again - and the sample then reads past it and faults + * the texture cache. Remember the largest extent described + * here so the fallback object can cover it. */ + if (unit < SGX_MAX_TEX_UNITS) { + /* The whole object, not the base level: a descriptor + * names a mip chain, six cube faces or a volume's + * slices at this address and the unit reads all of + * them. The chain runs to one texel however many + * levels the view reports, because that count is what + * it exposes rather than what the resource holds. + * height * stride alone left the object a third + * short and the cache faulted at 0x82555000, the + * base plus the tail of a 512x512 chain. Erring + * large here costs a third of one texture. */ + uint64_t need = 0; + uint32_t h = v->height, w = v->stride; + for (;;) { + need += (uint64_t)(h ? h : 1u) * (w ? w : 1u); + if (h <= 1u && w <= 1u) + break; + h >>= 1; + w >>= 1; + } + if (v->cube) + need *= 6u; + if (v->depth > 1u) + need *= v->depth; + need += v->border_map; + if (need > ctx->tex_need[unit]) + ctx->tex_need[unit] = need; + } + } + /* Recount rather than track: unbinding the highest unit has to shrink + * it, and a stale count submits a unit with no view. */ + ctx->nviews = 0; + for (unsigned i = 0; i < SGX_MAX_SAMPLERS; i++) + if (ctx->view_bound & (1u << i)) + ctx->nviews = i + 1; + /* The address a shader-issued sample reads changes when the view is + * bound, not when the frame is built. */ + ctx_refresh_sampler_state(ctx); + return 0; +} + +int sgx_bind_sampler_state(struct sgx_context *ctx, unsigned unit, + const struct sgx_sampler_state *st) +{ + if (!ctx || unit >= SGX_MAX_SAMPLERS) + return -EINVAL; + if (st) + ctx->samplers[unit] = *st; + else + memset(&ctx->samplers[unit], 0, sizeof(ctx->samplers[unit])); + /* The wrap and filter a shader-issued sample uses are in the block + * too, so a filter bound after the view has to reach it - without + * this, nearest and linear rendered identically. */ + ctx_refresh_sampler_state(ctx); + return 0; +} + +/* Write each bound unit's pair of words into the heap. They are the primary + * PDS program's own ds0 slots, which are not a fixed stride apart past the + * third unit, so xpsb_texctl_dw() computes them. */ +/* The depth and stencil state, into the frame's per-draw ISP word. It was + * being computed on every bind and then not written anywhere, so the depth + * test was whatever the captured frame had - on, with LESS - however the + * caller had set it. */ +/* ISP word A [27:25] is a three-bit object type, not a flag: 0 opaque, 1 + * read-modify-write, and the vendor uses 4 for punch-through. XPSB_ISP_BLEND + * is that field's low bit, so it has to be written as a field - or-ing a + * constant in is only correct while the value cannot exceed one. */ +/* Bits [27:25] are the ISP's PASS type - how it schedules the object - not + * its object type, which is [18:15] and says what shape it is. The DDK names + * them EURASIA_ISPA_PASSTYPE_* and EURASIA_ISPA_OBJTYPE_*; this was called + * the object type until the second one was needed for points and lines, and + * the two are unrelated fields. */ +#define XPSB_ISP_PASSTYPE_MASK (7u << 25) +#define SGX_ISP_TAGWRITEDIS (1u << 21) +#define SGX_ISP_PASS_OPAQUE 0u +#define SGX_ISP_PASS_TRANS 1u +#define SGX_ISP_PASS_TRANSPT 2u +#define SGX_ISP_PASS_FASTPT 4u + +/* The VDM command word for a draw of n indices. The primitive type has to + * match the ISP object type the same frame carries: the VDM decides how many + * indices make a primitive, the ISP what shape to rasterise from them, and a + * disagreement between the two is geometry read at the wrong stride. */ +/* Whether one frame may rasterise more than one shape. + * + * It can: a record carries its own ISP object type and its own VDM command, + * so the two agree per block rather than per frame. Off by default until it + * has been measured against the suite. */ +int sgx_mix_prim(void) +{ + const char *e = SGX_ENVS("SGX_MIX_PRIM"); + + return e && *e != '0'; +} + +static uint32_t sgx_ctx_draw_cmd_for(const struct sgx_context *ctx, + unsigned objtype, uint32_t n) +{ + uint32_t t = objtype == SGX_ISP_OBJ_LINE ? XPSB_VDM_LINES : + objtype == SGX_ISP_OBJ_TRI ? XPSB_VDM_TRIS : + XPSB_VDM_POINTS; + uint32_t cmd = XPSB_DRAW_CMD_TAG | XPSB_VDM_TYPE(t) | n; + const char *e = SGX_ENVS("SGX_VDM_IDXPRES"); + + /* The index presence bits, [25:22], which the captured tag carries set + * for the triangle it was capturing. They say how many indices the VDM + * takes for a primitive, so a point - which takes one - has no business + * asking for a third. SGX_VDM_IDXPRES replaces them to find out which + * the part wants. */ + if (e && *e) { + cmd &= ~XPSB_VDM_IDXPRES_MASK; + cmd |= ((uint32_t)strtoul(e, NULL, 0) << 22) & + XPSB_VDM_IDXPRES_MASK; + } + return cmd; +} + +/* The frame's own, for the geometry that is not a record's: the built-in + * triangle and the clearing quad both take whatever shape the frame is on. */ +static uint32_t sgx_ctx_draw_cmd(const struct sgx_context *ctx, uint32_t n) +{ + return sgx_ctx_draw_cmd_for(ctx, ctx->prim_objtype, n); +} + +/* What width the ISP should draw a line at, in pixels. + * + * This field sizes lines and nothing else. The vendor writes it from + * sState.sLine.fWidth alone - validate.c:3511, the non-545 branch, as + * (width + 0.5) - 1 in four bits - and a point's size never reaches the ISP + * at all: it is a vertex program output, FFGEN_OUTPUT_POINTSIZE, which the + * MTE reads at EURASIA_MTE_OUTPUT_OFFSET_POINTSIZE. So a sized point needs + * the vertex stage to emit it, which this driver does not do yet, and + * widening a point here would only write a field the part ignores. */ +static unsigned sgx_ctx_prim_width(const struct sgx_context *ctx) +{ + float w = ctx->line_width; + + /* SGX_POINT_WIDTH asks the same field for a sprite, which the vendor + * never does. It is here to tell a point that rasterises nothing + * because it has no size from one that rasterises nothing for another + * reason - a sprite may want the reserved coordinate hole before it + * draws at all. */ + if (ctx->prim_objtype == SGX_ISP_OBJ_SPRITEUV) { + const char *e = SGX_ENVS("SGX_POINT_WIDTH"); + + return e && *e ? (unsigned)strtoul(e, NULL, 0) : 1u; + } + if (ctx->prim_objtype != SGX_ISP_OBJ_LINE || !(w >= 1.0f)) + return 1u; + return (unsigned)(w + 0.5f); +} + +static uint32_t sgx_isp_passtype(uint32_t w, unsigned type) +{ + return (w & ~XPSB_ISP_PASSTYPE_MASK) | ((type & 7u) << 25); +} + +/* The pass type the ISP sorts an object by, which is what decides whether the + * part may remove it before shading. The vendor picks it in validate.c: a + * translucent blend is TRANS, and a program that can discard is punch-through + * - TRANSPT when it also blends, FASTPT when it does not. Discard was being + * sent as TRANS here, which asks for no hidden surface removal at all and + * shades every covered fragment of every alpha-tested surface. */ +static unsigned ctx_isp_passtype(const struct sgx_context *ctx) +{ + int kill = ctx->fs && ctx->fs->uses_kill; + int trans = ctx->blend_translucent; + + if (kill) + return trans ? SGX_ISP_PASS_TRANSPT : SGX_ISP_PASS_FASTPT; + return trans ? SGX_ISP_PASS_TRANS : SGX_ISP_PASS_OPAQUE; +} + +/* Both ISP fields at once: the pass type says how the object is scheduled, + * the object type what shape is rasterised from its vertices. */ +static uint32_t ctx_isp_word_a(const struct sgx_context *ctx, uint32_t a) +{ + unsigned type = ctx_isp_passtype(ctx); + uint32_t w = sgx_isp_objtype(sgx_isp_passtype(a, type), + ctx->prim_objtype, + sgx_ctx_prim_width(ctx)); + + /* No channel written is no pixel shaded: the vendor's ColorMask of + * zero (validate.c:3624-3631). Punch-through still has to reach the + * TSP to resolve its own coverage, so it wins below. */ + if (!ctx->colormask) + w |= SGX_ISP_TAGWRITEDIS; + if (type == SGX_ISP_PASS_TRANSPT || type == SGX_ISP_PASS_FASTPT) + w &= ~SGX_ISP_TAGWRITEDIS; + return w; +} + +/* The ISP state a draw opens its record with. */ +static void ctx_isp_words(const struct sgx_context *ctx, + struct sgx_isp_set *ff, struct sgx_isp_set *bf) +{ + *ff = ctx->isp_ff; + *bf = ctx->isp_bf; + ff->a = ctx_isp_word_a(ctx, ff->a); + if (ff->a & SGX_ISP_2SIDED) + bf->a = ctx_isp_word_a(ctx, bf->a); + /* Per record, not per frame: a query's draws carry the counter and the + * ones around them do not. */ + (void)sgx_isp_set_vistest(ff, bf, ctx->vis_reg); +} + +/* The vendor sends word B with every object - "HW doesn't have default for + * colormask" (validate.c:3884-3885) - so a frame in which any record carries + * one gives every record one, the default for those that had nothing to say. + * A frame with none keeps the captured block, which never had it. */ +static int ctx_frame_has_b(const struct sgx_context *ctx) +{ + unsigned k; + + for (k = 0; k < ctx->nrange; k++) + if (ctx->range[k].isp.a & SGX_ISP_BPRES) + return 1; + return 0; +} + +static void ctx_isp_for_record(const struct sgx_context *ctx, unsigned k, + struct sgx_isp_set *ff, struct sgx_isp_set *bf) +{ + *ff = ctx->range[k].isp; + *bf = ctx->range[k].isp_bf; + if (ctx_frame_has_b(ctx) && !(ff->a & SGX_ISP_BPRES)) { + ff->a |= SGX_ISP_BPRES; + ff->b = SGX_ISPB_DEFAULT; + } +} + +/* The MTE control word a draw opens its record with, and whether the group + * goes in at all. It used to be the frame's, written once at flush from + * whatever rasterizer state was bound last - so a frame whose draws culled + * different faces culled them all by the last one's: stencilcull's + * front-facing quad, drawn under glCullFace(GL_BACK), vanished because the + * draw after it culled the front. */ +static void ctx_cull_word(const struct sgx_context *ctx, uint32_t *rw, int *on) +{ + *rw = ctx->raster_word; + *on = *rw != 0; + /* Under hardware transform the vertices have NOT been transformed on + * the CPU, so bit 16 has to go - and the group has to be there at + * all, because its presence is what puts the MTE's viewport in + * circuit. A word of zero would otherwise leave the group out and the + * transform undone. */ + if (ctx->hwtcl && sgx_mte_viewport()) { + *rw &= ~(1u << 16); + *on = 1; + } +} + +static int ctx_emit_state(struct sgx_context *ctx) +{ + uint32_t *heap; + struct sgx_isp_set ff, bf; + int ret; + + if (!ctx->bo[SGX_CTX_BO_HEAP].map) + return -EINVAL; + heap = ctx->bo[SGX_CTX_BO_HEAP].map; + /* The block may move anywhere in its window. */ + if (XPSB_HEAP_STATE_END * 4 > ctx->bo[SGX_CTX_BO_HEAP].size) + return -ENOSPC; + /* The frame's own, which the first record uses; the rest carry theirs + * in a block of their own. */ + if (ctx->nrange && ctx->range) + ctx_isp_for_record(ctx, 0, &ff, &bf); + else + ctx_isp_words(ctx, &ff, &bf); + ret = xpsb_heap_set_isp(heap, &ff.a, &bf.a); + if (ret) + return ret; + /* The cull and shade-model word was being computed and then dropped, + * so the frame rasterised with whatever the captured block held - + * which carries no cull group at all, meaning no culling whatever the + * caller asked for. */ + /* On by default now. The group's position in the block was right all + * along; what was missing was bit 16 of the word - the vertices + * arrived already transformed - without which every triangle was + * discarded whichever winding was selected. SGX_NO_CULL puts it back + * to ignoring glCullFace, which is what this did before. */ + ctx_set_mte_viewport(ctx); + if (!SGX_ENV("SGX_NO_CULL")) { + uint32_t rw; + int on; + + /* Record 0's, like the ISP words: the frame's own block is + * what the first record copies. */ + if (ctx->nrange && ctx->range) { + rw = ctx->range[0].cull; + on = ctx->range[0].cull_on; + } else { + ctx_cull_word(ctx, &rw, &on); + } + ret = xpsb_heap_state_set(heap, XPSB_STATE_CULL, on, &rw); + if (ret) + return ret; + } + /* Where the block ended up, for the descriptor that DMAs it. The + * relocation writes the same address from cfg.state_off. */ + heap[XPSB_HEAP_STATE_DESC] = + (uint32_t)(ctx->bo[SGX_CTX_BO_HEAP].gpu_va + + xpsb_heap_state_base(heap) * 4u); + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: state block at %04x, mask %08x, %u " + "dwords, ctl %08x; isp %08x/%08x/%08x bf %08x/%08x/%08x\n", + xpsb_heap_state_base(heap), + heap[xpsb_heap_state_base(heap)], + xpsb_state_dwords(heap[xpsb_heap_state_base(heap)]), + heap[XPSB_HEAP_STATE_DESC + 1], + ff.a, ff.b, ff.c, bf.a, bf.b, bf.c); + return 0; +} + +/* The program the frame itself runs. Records past the first carry their own, + * built by ctx_put_program(); everything the frame configures - the iterator, + * the texture issues, the code in the slot and the secondary program that + * feeds it - has to describe this one. */ +static const struct sgx_shader *sgx_frame_fs(const struct sgx_context *ctx) +{ + if (ctx->nrange && ctx->range[0].fs) + return ctx->range[0].fs; + return ctx->fs; +} + +/* Which issue of the primary program a texture unit rides on, times the two + * dwords each issue takes in the PDS data. + * + * The unit is issue u only while every issue samples. A frame that iterates a + * varying has an issue that does not, so the unit sits one further along and + * its control, format and address live two dwords further into the block. + * Patching the fixed offsets left the sample reading the template's address - + * the context's own texture, ninety-six kilobytes, described as the caller's + * 512x512 - and the render faulted just past the end of it. */ +/* Which dwords of a primary PDS program carry a unit's texture state. The + * issue list decides it, so it has to be read off the program that record + * actually runs - a record carries its own now, and taking the frame's here + * patched the descriptor into another program's slots. */ +static unsigned ctx_tex_ds_fs(const struct sgx_shader *fs, unsigned unit) +{ + unsigned char issue_of[XPSB_MAX_TEX]; + unsigned nissue = 0, nu; + + /* The issue itself. Its three slots are ds0[2 + 2i], ds0[3 + 2i] and + * ds1[2 + 2i], and the data segment's banks alternate every eight + * dwords, so they are not a fixed stride apart once the fourth issue + * is reached - ds1[8] is memory dword 24, not 16. Each caller asks + * for the one it wants rather than stepping from a base. */ + if (!fs || !fs->attribs.nset) + return unit; + memset(issue_of, 0, sizeof issue_of); + nu = xpsb_attrib_issues(&fs->attribs, &nissue, issue_of); + if (!nu || unit >= nu || unit >= XPSB_MAX_TEX) + return unit; + return (unsigned)issue_of[unit]; +} + +static unsigned ctx_tex_ds(const struct sgx_context *ctx, unsigned unit) +{ + return ctx_tex_ds_fs(sgx_frame_fs(ctx), unit); +} + +/* Group 8 of the MTE state block: six floats, offset then scale per axis. + * xpsb_gen_heap() fills them from the render size, which is only right when + * the viewport covers the whole target. */ + +int sgx_set_viewport(struct sgx_context *ctx, const float *scale, + const float *translate) +{ + unsigned i; + + if (!ctx || !scale || !translate) + return -EINVAL; + for (i = 0; i < 3; i++) { + ctx->vp_scale[i] = scale[i]; + ctx->vp_translate[i] = translate[i]; + } + ctx->vp_set = 1; + return 0; +} + +/* Write the caller's viewport into group 8. Only under the MTE scheme: on the + * epilogue path the group is inert and the vertex program reads the viewport + * out of the record instead. */ +/* Into one state block, with one answer about who transformed the geometry: + * a frame may hold records of both kinds and group 8 belongs to the record, + * not to the frame. */ +static void ctx_mte_viewport_into(struct sgx_context *ctx, uint32_t *heap, + int hwtcl) +{ + uint32_t *g8; + unsigned i; + + if (!heap || !ctx->vp_set || !sgx_mte_viewport()) + return; + g8 = heap + xpsb_heap_state_off(heap, XPSB_STATE_VIEWPORT); + for (i = 0; i < 3; i++) { + float t = ctx->vp_translate[i], sc = ctx->vp_scale[i]; + + /* Identity when the part is not running the transform. The + * draw module applies the viewport itself and hands over + * window coordinates, so leaving the caller's viewport in + * group 8 applies it a second time and the geometry lands off + * the target. twm's menu showed it exactly: the background + * quads went to the part and appeared, every glyph went to + * the draw module as points - the X server draws core-font + * text that way - and vanished, as did the separator lines. + * The damage shift below is already gated this way; group 8 + * itself was not. */ + if (!hwtcl) { + t = 0.0f; + sc = 1.0f; + } + /* A damaged render draws the damage rectangle at the target's + * origin, so the window coordinate the MTE produces has to + * move with it - which on this scheme is the translate, the + * last thing applied, rather than the record dword the + * epilogue scheme shifts (ctx_upload_vertices). */ + if (hwtcl && ctx->dmg_on > 0 && i == 0) + t -= (float)ctx->dmg_x0; + if (hwtcl && ctx->dmg_on > 0 && i == 1) + t -= (float)ctx->dmg_y0; + memcpy(&g8[i * 2u], &t, 4); + memcpy(&g8[i * 2u + 1u], &sc, 4); + } + /* What was written, not what was asked for: the two differ on the + * draw module's path and reporting the caller's numbers hid that. */ + if (SGX_ENV("SGX_DEBUG")) { + float w[6]; + + for (i = 0; i < 6; i++) + memcpy(&w[i], &g8[i], 4); + fprintf(sgx_log(), "sgx: mte viewport: %g+%g %g+%g %g+%g " + "(hwtcl %d)\n", w[0], w[1], w[2], w[3], w[4], w[5], + hwtcl); + } +} + +static void ctx_set_mte_viewport(struct sgx_context *ctx) +{ + uint32_t *heap = ctx->bo[SGX_CTX_BO_HEAP].map; + + if (!heap || XPSB_HEAP_STATE_END * 4u > ctx->bo[SGX_CTX_BO_HEAP].size) + return; + ctx_mte_viewport_into(ctx, heap, ctx->hwtcl); +} + +/* The record the part's transform writes and the frame's iterator reads. + * + * sgx_hwtcl_record_dwords() sizes it at three floats a set, which is what a + * sampled set carries, but the vertex program writes its outputs into o[] four + * apart - sgx_vs_out_slots() puts slot k at 4 * k - so a second set was written + * at the fourth float and read at the third. Off by one register, which is a + * varying arriving with the neighbouring set's value or with its own fourth + * component zero. The fragment program's own stride is the one both agree on. */ +static unsigned ctx_hwtcl_record_floats(const struct sgx_context *ctx, + unsigned uni_dwords) +{ + unsigned w = sgx_hwtcl_record_dwords(uni_dwords); + const struct sgx_shader *fs = sgx_frame_fs(ctx); + unsigned need; + + if (!fs || !fs->attribs.nset) + return w; + need = xpsb_attrib_stride(&fs->attribs); + return need > w ? need : w; +} + +static int ctx_emit_samplers(struct sgx_context *ctx, + const struct sgx_shader *fs) +{ + uint32_t *heap; + /* The captured slot, always: the builder copies these into whatever + * region it lays the program out in, so this stays one fixed place + * rather than a second guess at which one that will be. */ + unsigned base = XPSB_PRI_PDS_OFF; + unsigned i; + + /* Not only when something is sampled: an untextured program that reads + * a varying still needs the frame's iterator configured for it, and + * this is where that is written. Returning early when neither applied + * left every unit as a descriptor of zeros, which is the address-zero + * fault the loop below exists to prevent - so describe the units + * first and only then decide there is nothing more to do. */ + if (!ctx->bo[SGX_CTX_BO_HEAP].map) + return -EINVAL; + heap = ctx->bo[SGX_CTX_BO_HEAP].map; + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: emit samplers: fs %p nset %u nprimattr " + "%u preiter %u hwtcl %d\n", (const void *)fs, + fs ? fs->attribs.nset : 0, fs ? fs->nprimattr : 0, + fs ? fs->tex_preiterated : 0, ctx->hwtcl); + + /* Every unit, not only the bound ones. xpsb_heap_set_attribs() clears + * this region and reads these two words back to build the primary PDS + * data, so a unit left undescribed goes to the hardware as a + * descriptor of zeros - which the render reads at address zero, and + * the core stops with the texture cache named as the requestor. A unit + * with no caller texture describes the context's own, which is + * allocated at a fixed address for exactly this. */ + for (i = 0; i < SGX_MAX_TEX_UNITS; i++) { + /* The context keeps its own buffer for the two units the + * captured frame carries. A unit past those has none and falls + * through to the caller's object or to the heap below, which + * is all a descriptor nothing samples needs - "i ? TEX2 : TEX" + * gave every unit above the first the second one's buffer. */ + int slot = i == 0 ? (int)SGX_CTX_BO_TEX : + i == 1 ? (int)SGX_CTX_BO_TEX2 : -1; + const struct sgx_sampler_view *v = &ctx->views[i]; + const struct sgx_sampler_state *st = &ctx->samplers[i]; + struct sgx_sampler_state dst; + struct sgx_sampler_view dv; + uint32_t w0 = 0, w1 = 0; + uint64_t at = xpsb_texctl_dw_at(xpsb_tex_unit_base(i), i); + + if (i >= ctx->nviews || !(ctx->view_bound & (1u << i)) || + !ctx->views[i].gpu_va) { + uint64_t va = slot >= 0 ? ctx->bo[slot].gpu_va : 0; + unsigned dim = SGX_CTX_TEX_DIM; + + /* The context's own buffer is not there while a + * caller's texture holds the unit's address, and the + * unit still has to be described: leaving it at zeros + * is what made the render read address zero and the + * core stop with the texture cache as the requestor. + * Describe what is mapped there instead, at a size + * that fits inside it - nothing samples it. */ + if (!va && ctx->tex_bo[i]) { + va = ctx->tex_bo[i]->gpu_va; + dim = 32u; + } + /* Last resort: the heap is always mapped. Skipping + * instead left the unit as a descriptor of zeros, + * which is the address-zero fault with the texture + * cache as the requestor - and with a texture kept at + * its own address the context's own buffer is not + * bound, so this is the ordinary case there. */ + if (!va) { + va = ctx->bo[SGX_CTX_BO_HEAP].gpu_va; + dim = 32u; + } + if (!va) + continue; + memset(&dv, 0, sizeof dv); + memset(&dst, 0, sizeof dst); + dv.gpu_va = (uint32_t)va; + dv.width = dim; + dv.height = dim == SGX_CTX_TEX_DIM ? dim : 1u; + dv.stride = dim * 4u; + dv.format = SGX_FMT_A8R8G8B8; + dv.nlevels = 1; + v = &dv; + st = &dst; + } + if (sgx_sampler_words(v, st, &w0, &w1) != SGX_SAMPLER_OK) + return -EINVAL; + if ((at + 2) * 4 > ctx->bo[SGX_CTX_BO_HEAP].size) + return -ENOSPC; + heap[at] = w0; + heap[at + 1] = w1; + ctx->tex_emitted++; + /* What the hardware will actually fetch, read back through the + * object's own map: a descriptor that names the right address + * says nothing about what is in the memory at it. */ + if (SGX_ENV("SGX_DUMP_TEXELS") && i == 0 && + ctx->bo[SGX_CTX_BO_TEX].map) + fprintf(sgx_log(), "sgx: ctx own TEX at 0x%llx: " + "%08x %08x %08x %08x\n", + (unsigned long long)ctx->bo[SGX_CTX_BO_TEX].gpu_va, + ((const uint32_t *)ctx->bo[SGX_CTX_BO_TEX].map)[0], + ((const uint32_t *)ctx->bo[SGX_CTX_BO_TEX].map)[1], + ((const uint32_t *)ctx->bo[SGX_CTX_BO_TEX].map)[2], + ((const uint32_t *)ctx->bo[SGX_CTX_BO_TEX].map)[3]); + if (SGX_ENV("SGX_DUMP_TEXELS") && ctx->tex_bo[i] && + ctx->tex_bo[i]->map) + fprintf(sgx_log(), "sgx: unit %u texels at 0x%llx: " + "%08x %08x %08x %08x\n", i, + (unsigned long long)ctx->tex_bo[i]->gpu_va, + ((const uint32_t *)ctx->tex_bo[i]->map)[0], + ((const uint32_t *)ctx->tex_bo[i]->map)[1], + ((const uint32_t *)ctx->tex_bo[i]->map)[2], + ((const uint32_t *)ctx->tex_bo[i]->map)[3]); + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: unit %u descriptor %08x %08x " + "(va 0x%08x, %s)\n", i, w0, w1, + (unsigned)v->gpu_va, + v == &dv ? "the context's own" : "a view"); + } + + if (SGX_ENV("SGX_DUMP_PDS")) { + const uint32_t *d = (const uint32_t *)((const char *)heap + + XPSB_PRI_PDS_OFF); + unsigned q; + + fprintf(sgx_log(), "sgx: frame pds:"); + for (q = 0; q < 32; q++) + fprintf(sgx_log(), " %s%08x", q % 4 ? "" : "| ", d[q]); + fprintf(sgx_log(), "\n"); + } + + /* The words above are not what the hardware reads. The primary PDS + * program carries a copy of them, baked in when it is generated, and + * the texture unit is fed from that - so a descriptor written into the + * heap and left there is inert, which is exactly what it looked like: + * the size, the wrap mode and the filter all changed and not one pixel + * moved. Rebuilding the program is what makes them take. */ + /* The bound fragment program says what has to be iterated: the packed + * colour and one texel per sampled unit as before, or - for a program + * that computes with a varying - the varying's own coordinate set, + * delivered as floats. + * + * This describes what the vertex program emits and what the fragment + * task is fed, which is the same on both paths. What the hardware path + * changes is only the record it fetches: that carries the viewport on + * top of the attributes, so it is wider, and it is re-sized below. + * Driving both from a fixed two-unit shape left the fragment task + * waiting on texture issues its program never asked for. */ + ctx->pri_pds_dwords = 0; + { + struct xpsb_attribs hw; + struct xpsb_attribs spr; + const struct xpsb_attribs *a = fs ? &fs->attribs : NULL; + + /* The sprite's own coordinate, which the ISP produces and the shader + * never writes. Every point object type this part has is a "UV" form, + * so the coordinate is produced whether or not anything reads it, and + * the set has to be there for it. SGX_NO_SPRITE_UV drops it. */ + if (sgx_sprite_uv_set(ctx)) + a = ctx_sprite_attribs(a, &spr); + + /* Under the part's transform the varyings are the vertex program's + * outputs past the position, four dwords each, whatever coordinate + * sets the fragment program declared - a program reading only an + * interpolated colour declares none, so the frame kept the template's + * iterators and the fragment task waited on ones its program never + * asked for. Only when the program reads primary attributes at all: a + * flat-colour program reads none, and describing sets for it stopped + * the fans and the mat4 cases rendering anything. */ + /* Slot 1 is the record's colour quad, not a coordinate set: + * sgx_vs_out_slots() puts a colour at SGX_VTX_COLOUR_DW and the sets + * begin at SGX_VTX_SET_DW behind it, so the sets are the outputs above + * slot 1 and nvtxout counts from slot 0. Taking nvtxout - 1 counted + * the colour as a set: a fixed-function transform, which writes + * position and a colour and nothing else, declared one iterated set at + * dword eight and emitted only dwords nought to seven, and the + * fragment task waited for an iteration the vertex stage never sent - + * every such frame lost with the tiling engine stopped and no MMU + * fault. That is glxgears, and any lit or per-vertex-coloured + * fixed-function draw. */ + if (ctx->hwtcl && fs && fs->nprimattr && ctx->vs && + ctx->vs->nvtxout > 2 && (!a || !a->nset)) { + unsigned n = ctx->vs->nvtxout - 2u; + + if (n > XPSB_NCOORD_MAX) + n = XPSB_NCOORD_MAX; + memset(&hw, 0, sizeof hw); + hw.nset = n; + for (i = 0; i < n; i++) { + hw.set[i].width = 4; + hw.set[i].iterated = 1; + } + a = &hw; + } + /* Under the part's transform every set is four floats, because that is + * what the transform writes: sgx_vs_out_slots() gives each output a + * slot and the backend emits it at QUAD * slot. The frame declares the + * widths in group 14 - attrib_tcset_word() - and derives the record + * stride and the iterator's bases from the same numbers, so declaring + * a sampled set at three floats where the transform wrote four left + * every one of them disagreeing with the vertex. Normalising here + * rather than at each of those four means none of them needs to know + * which stage filled the record. */ + /* That reason is gone. The vertex program is given the iterator's own + * packed offsets now - sgx_shader_vtx_layout_dw() - so it writes each + * set where the set actually is, and widening them all to four makes + * the frame describe something the transform did not write. Measured + * on ioquake3: the same draw came out as group 14 0x3f and a sixteen + * dword record under the transform against 0x3d and fifteen without + * it, which is one set described a float wider than it is. + * + * SGX_NO_PACKED_VSOUT puts both halves back together. */ + if (ctx->hwtcl && a && a->nset && SGX_ENVS("SGX_NO_PACKED_VSOUT")) { + unsigned q; + + if (a != &hw) + hw = *a; + for (q = 0; q < hw.nset && q < XPSB_NSET_MAX; q++) + hw.set[q].width = 4; + a = &hw; + } + /* A program whose issue list will not fit the primary PDS region left + * this block unentered and said nothing: the frame kept the captured + * template's iterators, which describe a different program, and the + * draw rendered black. The region is XPSB_PRI_PDS_OFF to + * XPSB_PDS_VTXDESC_OFF - twenty-four dwords for the data and the code + * together - and three texture issues do not fit in it, which is what + * caps this driver at two texture units. Said once, with the shape + * that did not fit. */ + /* Every program, including one with no coordinate set: the frame + * carries one configuration and it used to be left as the previous + * program installed it when this one had no sets to install. A plain + * quad after the polygon stipple's variant then ran with that + * program's set, its texture issue and its punch-through dependency + * in the primary PDS, against an opaque ISP word and an eleven-float + * record - and the core stalled on it. attrib_issues() gives a setless + * program the TAG issue a textureless task needs and nothing else. */ + if (a && xpsb_heap_set_attribs(heap, a, &ctx->pri_pds_dwords)) { + static int said; + + if (!said) { + unsigned q, nsamp = 0; + + said = 1; + for (q = 0; q < a->nset && q < XPSB_NSET_MAX; q++) + nsamp += a->set[q].sampled; + fprintf(sgx_log(), "sgx: the primary PDS program for %u " + "set(s) and %u sampled unit(s) does not fit " + "the frame's region - the frame keeps the " + "template's iterators and the draw will not " + "match what was asked for\n", a->nset, nsamp); + } + } + ctx->pri_pds_off = a ? xpsb_pri_pds_off(a) : XPSB_PRI_PDS_OFF; + if (a && !xpsb_heap_set_attribs(heap, a, &ctx->pri_pds_dwords)) { + /* A setless program's record is not the sets' stride: it is + * the eleven floats the vbuf always writes (sgx_update_vtx_ + * floats), so that width goes back over the eight the sets + * alone add up to. */ + if (a->nset) + ctx->vtx_floats = xpsb_attrib_stride(a) + + sgx_record_extra_dwords(ctx); + else + xpsb_heap_set_record_width(heap, ctx->vtx_floats, 0); + /* xpsb_heap_set_attribs() programs the stride from the sets + * alone, which know nothing about a point size riding at the + * end. Without this the heap says eleven floats while the + * records are twelve, and the fetch walks the buffer at the + * wrong stride - the widened point came out in the corner of + * the target rather than where it was drawn. */ + if (sgx_record_extra_dwords(ctx)) + xpsb_heap_set_record_width(heap, ctx->vtx_floats, 0); + /* The width the records were actually written at wins. This + * is computed from whatever program is bound when the frame + * is flushed, which is the last one the caller happened to + * bind - and the vertices already in the buffer were written + * at the width their own program asked for. Programming the + * other one makes the fetch walk the buffer at the wrong + * stride, so every vertex after the first comes out further + * from where it belongs and the triangles rasterise as long + * thin wedges: a scrolled list drawn as streaks. */ + /* Not under hardware transform: there the record is the wider + * of this stride and what the transform emits, so the stride + * alone is provisional and comparing it reported a mismatch + * on every glmark2 build frame that was rendering correctly. + * The branch below owns the width on that path. */ + /* The record width is three fields outside the per-record + * window, so it is the one thing a per-record vertex PDS copy + * does not carry - and it belongs to the vertices in the + * buffer, not to whatever program happens to be bound now. + * Binding the next operation's program lays it out and moves + * vtx_floats before the frame is flushed, so a frame holding + * eleven-float records was fetched at twelve: X's first fill + * of every client after the first went missing. The hardware + * transform branch below has always had this rule; the + * software path had it behind an environment variable. */ + if (!ctx->hwtcl && ctx->frame_vtx_floats && + ctx->frame_vtx_floats != ctx->vtx_floats && + !SGX_ENV("SGX_NO_FRAME_STRIDE")) { + /* Behind SGX_DEBUG now that this is the rule rather + * than the exception: any bind after the frame's last + * draw reaches it, which under glamor is most frames. */ + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: attribs: the records are " + "%u floats, not the %u this program " + "asks for\n", ctx->frame_vtx_floats, + ctx->vtx_floats); + ctx->vtx_floats = ctx->frame_vtx_floats; + xpsb_heap_set_record_width(heap, ctx->vtx_floats, 0); + } + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: attribs: %u set(s), stride %u, " + "regs %u, g10 %08x, g14 %08x, itr %08x\n", + a->nset, ctx->vtx_floats, + heap[xpsb_heap_pds_dw(heap, XPSB_PDS_W_CTL)] & + XPSB_VARYING_MASK, + heap[xpsb_heap_state_off(heap, XPSB_STATE_OUTSEL)], + heap[xpsb_heap_state_off(heap, XPSB_STATE_TEXSIZE)], + heap[XPSB_PRI_PDS_DW + 8 + 1]); + if (SGX_ENV("SGX_DEBUG") && a) { + unsigned q; + + for (q = 0; q < a->nset; q++) + fprintf(sgx_log(), "sgx: set %u: width %u " + "sampled %u values %u colour %u " + "proj %u f16 %u iterated %u\n", q, + a->set[q].width, a->set[q].sampled, + a->set[q].values, a->set[q].on_colour, + a->set[q].projected, a->set[q].f16, + a->set[q].iterated); + } + } else { + /* Silent until now, and a refused set is not the same as a + * program with none: the frame then keeps the template's + * iterators and the fragment task waits on issues its program + * never asks for. */ + unsigned n = ctx->hwtcl ? (unsigned)sgx_hwtcl_ntex() : + ctx->nviews; + + if (fs && fs->attribs.nset && SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: attribs: %u set(s) refused; the " + "frame keeps the template's iterators\n", + fs->attribs.nset); + xpsb_heap_set_ntex(heap, n); + } + } + /* Whether the MTE should look for a point size at the end of the + * record. Out here rather than beside the sets: a program with no + * varyings at all - a point drawn in one flat colour is exactly that - + * never reaches that branch, so the bit was never set for the one kind + * of draw that needs it. Cleared as well as set, or it outlives the + * frame that wanted it and every later record is read a dword long. */ + if (ctx->point_size_in_record) { + heap[xpsb_heap_state_off(heap, XPSB_STATE_OUTSEL)] |= + XPSB_STATE_G10_SIZE; + xpsb_heap_set_record_width(heap, ctx->vtx_floats, 0); + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: point size in record: g10 " + "%08x, width %u, objtype %u\n", + heap[xpsb_heap_state_off(heap, XPSB_STATE_OUTSEL)], + ctx->vtx_floats, ctx->prim_objtype); + } else { + heap[xpsb_heap_state_off(heap, XPSB_STATE_OUTSEL)] &= + ~XPSB_STATE_G10_SIZE; + } + /* The fetch, last, because both branches above sized it for what the + * program emits rather than for what the record holds. */ + if (ctx->hwtcl) { + int r; + + ctx->vtx_floats = ctx_hwtcl_record_floats(ctx, ctx->vs_const_dwords); + /* And here too the width the records were written at wins. + * This branch sizes the fetch from the transform that is + * bound when the frame is flushed, while the vertices in the + * buffer were written by whatever was bound as each draw came + * in - and sgx_update_vtx_floats() leaves vtx_floats alone + * while hardware transform is on, so the draw-time check + * cannot see the difference either. Reading a frame at a + * stride its vertices were not written at walks the buffer + * askew, and the triangles come out as long thin wedges. */ + if (ctx->frame_vtx_floats && + ctx->frame_vtx_floats != ctx->vtx_floats && + !SGX_ENV("SGX_NO_FRAME_STRIDE")) { + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: hwtcl: the records are " + "%u floats, not the %u the transform " + "asks for\n", ctx->frame_vtx_floats, + ctx->vtx_floats); + ctx->vtx_floats = ctx->frame_vtx_floats; + } + r = sgx_hwtcl_set_record(heap, ctx->vtx_floats); + if (r) + return r; + /* Group 10 keeps the record's width here. Naming what the + * vertex program writes instead (nvtxout * 4) is the shape the + * field's name suggests, but it stops the flat-colour fans and + * the mat4 cases rendering, so it is not what the field means + * on this path. SGX_MTE_OUT_DW sets it to measure. */ + /* Written here rather than left to the width-mismatch path + * above, which reached it only through set_record_width() and + * only when the two widths happened to differ. */ + r = sgx_hwtcl_set_out_dwords(heap, ctx->vtx_floats); + if (r) + return r; + if (ctx->vs && ctx->vs->nvtxout && SGX_ENV("SGX_MTE_OUT_DW")) { + r = sgx_hwtcl_set_out_dwords(heap, + ctx->vs->nvtxout * 4u); + if (r) + return r; + } + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: hwtcl: record %u dword(s), the " + "vertex program writes %u\n", ctx->vtx_floats, + ctx->vs ? ctx->vs->nvtxout * 4u : 0); + } + { + const char *e = SGX_ENVS("SGX_HEAP_POKE_OFF"); + + if (e && *e) { + unsigned off = (unsigned)strtoul(e, NULL, 0); + const char *v = SGX_ENVS("SGX_HEAP_POKE_VAL"); + + heap[off / 4] ^= v ? (uint32_t)strtoul(v, NULL, 0) : 1u; + fprintf(sgx_log(), "sgx: poked heap[0x%x] -> %08x\n", + off, heap[off / 4]); + } + } + return 0; +} + +int sgx_use_target(struct sgx_context *ctx, struct sgx_bo *bo) +{ + int ret; + struct sgx_bo *prev = NULL; + uint64_t prev_va = 0; + unsigned u; + int alias = 0; + + if (!ctx || !bo) + return -EINVAL; + if (SGX_ENV("SGX_DUMP_CMD")) + fprintf(sgx_log(), "sgx: target bo %p size %llu (was %p)\n", + (void *)bo, (unsigned long long)bo->size, + (void *)ctx->target_bo); + if (ctx->target_bo == bo) + return 0; + /* If the incoming target already has an address in the target window, + * take it as it stands rather than moving it to the fixed one. The + * frame's relocations read the address out of the object below, so the + * fixed address is a convention rather than a requirement - and moving + * a screen-sized pixmap costs a page table entry and a clflush for + * every one of its pages (mmu.c, psb_mmu_clflush per entry), which a + * display server pays several times a frame as it ping-pongs between + * a window and the screen. + * + * Off by default, and not ready to be on: leaving every pixmap bound + * consumes the target window, and once it is full the binding thrashes. + * Measured over four menu posts on one server, in ms: 342, 3916, 3378, + * 3396 with it, against 824, 1876, 1853, 1860 without. Two and a half + * times quicker on a fresh server and twice as slow once it has been + * used, which is the wrong trade until the window is recycled - the + * first post alone would have sold it. SGX_TARGET_IN_PLACE asks for + * it. */ + if (SGX_ENVS("SGX_TARGET_IN_PLACE") && bo->gpu_va && + bo->window == sgx_ctx_window[SGX_CTX_BO_TARGET]) { + if (ctx->bound[SGX_CTX_BO_TARGET]) { + sgx_bo_free(ctx->ws, &ctx->bo[SGX_CTX_BO_TARGET]); + ctx->bound[SGX_CTX_BO_TARGET] = 0; + } + ctx->target_prev_va = bo->gpu_va; + ctx->target_bo = bo; + ctx->target_is_scanout = 0; + ctx->bo[SGX_CTX_BO_TARGET] = *bo; + return 0; + } + /* Whatever holds the target address gives it up first: the context's + * own allocation, or the object bound last time. The one bound last + * time goes back where it was, because a resource that was a render + * target is usually about to be a texture - that is what rendering to + * a texture means - and a sampler descriptor names its address. */ + if (ctx->bound[SGX_CTX_BO_TARGET]) { + sgx_bo_free(ctx->ws, &ctx->bo[SGX_CTX_BO_TARGET]); + ctx->bound[SGX_CTX_BO_TARGET] = 0; + } else if (ctx->target_bo && + ctx->target_bo->gpu_va == sgx_ctx_va[SGX_CTX_BO_TARGET]) { + /* Only if it is still here. A pixmap that was the target and + * has since become a texture sits at the unit's address now, + * and unbinding it from there took the texture's mapping with + * it - the render then faulted on the address the frame + * samples every draw. */ + sgx_bo_unbind(ctx->ws, ctx->target_bo); + prev = ctx->target_bo; + prev_va = ctx->target_prev_va; + } + /* glamor copies a pixmap onto itself when a window moves, so the + * incoming target can be the object a texture unit is already reading. + * Evicting it and putting the context's own texture back made the copy + * source the fallback checkerboard, which is what a moved window came + * out as. One buffer can only be at one address, so instead the unit + * follows it to the target's - the same buffer read and written, which + * is what a self-copy is. */ + for (u = 0; u < 2; u++) + if (ctx->tex_bo[u] == bo) + alias = 1; + ctx->target_prev_va = bo->gpu_va; + /* The outgoing target is about to be unbound and rebound elsewhere, + * and a deferred submit means the render into it may still be + * running: moving its mapping out from under the core loses whatever + * had not been written yet. That is a window's content when a display + * server moves a window and copies the old pixels out of what was the + * target a moment ago - without this the copy came back black. + * + * It used to be unconditional, and it cost: ioquake3 went from about + * 13 fps to 10, and a display server pays it on every pixmap it draws + * into - one twm menu switches target dozens of times, and each switch + * drained the core. + * + * Narrowing it to targets a texture unit already names does not work, + * because the view that will sample it is not bound until after the + * switch. So it is narrowed by buffer instead, which is the per-buffer + * busy flag this asked for: a frame marks what it names when it is + * submitted, and only a buffer that a fired render may still be using + * makes this wait. Everything else moves without one. + */ + /* The outgoing target being busy is the common case for a double + * buffered client, and waiting for it is what stops frame N+1 being + * built while frame N renders. It only has to move because the + * incoming one wants its address; give the incoming one the other slot + * and nothing moves, so nothing has to be waited for. Only when the + * target fits a slot, and only under SGX_TARGET_SLOTS - the frame then + * carries whichever address the object actually has, which it already + * reads from the object rather than from the table. */ + uint64_t want = sgx_ctx_va[SGX_CTX_BO_TARGET]; + int two_slots = sgx_target_slots() && + ctx->bo[SGX_CTX_BO_TARGET].size <= SGX_CTX_TARGET_SLOT; + + if (two_slots && ctx->target_bo && ctx->target_bo != bo && + sgx_bo_is_busy(ctx->ws, ctx->target_bo) && + ctx->target_bo->gpu_va == want) + want = SGX_CTX_TARGET_VA2; + + if (sgx_bo_is_busy(ctx->ws, bo) || + (want == sgx_ctx_va[SGX_CTX_BO_TARGET] && + (sgx_bo_is_busy(ctx->ws, ctx->target_bo) || + sgx_bo_is_busy(ctx->ws, prev)))) + sgx_wait_idle(ctx->ws); + if (bo->gpu_va) + sgx_bo_unbind(ctx->ws, bo); + ret = sgx_bo_bind_at(ctx->ws, bo, sgx_ctx_window[SGX_CTX_BO_TARGET], + want); + /* The fixed address is a convention, not a requirement: the frame's + * relocations read the target out of the object below, so anywhere in + * the window serves. Refusing the framebuffer when something else + * still holds 0x80000000 is what left glmark2's refract scene black - + * its render-to-texture pass was never installed and every draw went + * to the target before it. */ + if (ret) { + int any = sgx_bo_bind(ctx->ws, bo, + sgx_ctx_window[SGX_CTX_BO_TARGET]); + + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: target: 0x%llx refused (%d), " + "anywhere %d -> 0x%llx\n", + (unsigned long long)want, ret, any, + (unsigned long long)bo->gpu_va); + ret = any; + } + /* Back where it was if that address is still free, and anywhere if it + * is not - but only after the incoming target holds the fixed address, + * because "anywhere" is free to take that one otherwise. Ignoring the + * failure left the object with no address at all, and the next draw + * that named it faulted the vertex fetch at zero. */ + if (prev && + (!prev_va || sgx_bo_bind_at(ctx->ws, prev, + sgx_ctx_window[SGX_CTX_BO_TARGET], + prev_va)) && + sgx_bo_bind(ctx->ws, prev, sgx_ctx_window[SGX_CTX_BO_TARGET]) && + SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: the previous target kept no address; " + "a draw naming it will be refused\n"); + /* A view of the object that just stopped being the target still names + * the target's fixed address, which the incoming target now holds - + * so the sample read the surface being drawn into rather than the one + * drawn earlier. A display server that renders a menu into a pixmap + * and then copies it to the screen does exactly this, and copied the + * screen onto itself. The object has moved; the view has to follow. */ + if (prev && prev->gpu_va != sgx_ctx_va[SGX_CTX_BO_TARGET]) { + unsigned u; + + for (u = 0; u < SGX_MAX_TEX_UNITS; u++) { + enum sgx_ctx_bo sl = u ? SGX_CTX_BO_TEX2 : + SGX_CTX_BO_TEX; + + if ((ctx->tex_bo[u] == prev || + ctx->views[u].gpu_va == + (uint32_t)sgx_ctx_va[SGX_CTX_BO_TARGET]) && + ctx->views[u].gpu_va) { + ctx->views[u].gpu_va = (uint32_t)prev->gpu_va; + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: unit %u follows " + "the old target to 0x%llx\n", u, + (unsigned long long)prev->gpu_va); + } + /* And the unit's own copy of the object. It is a copy + * by value, so it keeps the address the object had + * when it was taken; left behind, the unit describes + * and unbinds an address the object no longer has. */ + if (ctx->tex_bo[u] == prev && !ctx->bound[sl] && + ctx->bo[sl].handle == prev->handle) + ctx->bo[sl].gpu_va = prev->gpu_va; + } + } + if (ret) + return ret; + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: bind target bo %p -> 0x%llx (tex0 bo %p " + "at 0x%llx)\n", (void *)bo, + (unsigned long long)bo->gpu_va, + (void *)ctx->tex_bo[0], + (unsigned long long)ctx->bo[SGX_CTX_BO_TEX].gpu_va); + if (alias) { + enum sgx_ctx_bo slot; + + for (u = 0; u < 2; u++) { + if (ctx->tex_bo[u] != bo) + continue; + slot = u ? SGX_CTX_BO_TEX2 : SGX_CTX_BO_TEX; + /* The slot stops being the context's own here, so the + * allocation it held has to go before the caller's + * buffer overwrites it - and the ownership flag with + * it. Left set, the context believed it owned the + * caller's pixmap: the next sgx_use_texture freed it, + * closing glamor's handle, and every later submit + * naming that pixmap was refused with ENOENT. That is + * the repaint of the area a menu or a moved window + * uncovered. */ + if (ctx->bound[slot]) { + sgx_bo_free(ctx->ws, &ctx->bo[slot]); + ctx->bound[slot] = 0; + } + ctx->bo[slot] = *bo; + ctx->tex_prev_va[u] = 0; + if (ctx->view_bound & (1u << u)) { + ctx->views[u].gpu_va = (uint32_t)bo->gpu_va; + ctx_refresh_sampler_state(ctx); + } + } + } + ctx->target_bo = bo; + ctx->target_is_scanout = 0; + /* The relocations read the target's address out of the frame's own + * object, so it carries the caller's. */ + ctx->bo[SGX_CTX_BO_TARGET] = *bo; + return 0; +} + +/* Give the target address back, so the next framebuffer is rendered into an + * allocation sized for it. + * + * sgx_set_framebuffer() allocates the context's own colour target only when no + * caller buffer holds the address, so a framebuffer with no colour attachment + * kept whatever was bound last. glmark2's shadow scene binds a 2048x1151 + * depth-only target over a 1280x720 window and the pixel back end then wrote + * colour past the end of it: an MMU fault named at the PBE, the frame lost and + * the core recovered, forty times in a five second run. */ +/* The frame's depth buffer is the caller's object. The relocations read the + * address out of ctx->bo[SGX_CTX_BO_DEPTH], the same way the target's is + * read out of its slot, so taking the slot is all this has to do. + * + * Rendering into a depth texture had no path at all before: set_framebuffer + * recorded that a depth attachment existed and nothing else, so the render + * wrote the context's own buffer at the depth window's address while the + * sample read the caller's resource somewhere else entirely. The measured + * shape of that is a depth texture that reads as zero everywhere - a shadow + * compare that answers the same thing for every pixel, whatever the + * comparison. */ +int sgx_use_depth(struct sgx_context *ctx, struct sgx_bo *bo, uint64_t need) +{ + struct sgx_bo *prev = NULL; + uint64_t prev_va = 0; + int ret; + + if (!ctx || !bo) + return -EINVAL; + if (ctx->depth_bo == bo) + return 0; + /* The raster pass writes the padded surface, so an object that only + * covers the visible rows is not one the frame may be pointed at. */ + if (bo->size < need) + return -ENOSPC; + if (ctx->bound[SGX_CTX_BO_DEPTH]) { + sgx_bo_free(ctx->ws, &ctx->bo[SGX_CTX_BO_DEPTH]); + ctx->bound[SGX_CTX_BO_DEPTH] = 0; + } else if (ctx->depth_bo && + ctx->depth_bo->gpu_va == sgx_ctx_va[SGX_CTX_BO_DEPTH]) { + prev = ctx->depth_bo; + prev_va = ctx->depth_prev_va; + if (sgx_bo_is_busy(ctx->ws, prev)) + sgx_wait_idle(ctx->ws); + sgx_bo_unbind(ctx->ws, prev); + } + if (sgx_bo_is_busy(ctx->ws, bo)) + sgx_wait_idle(ctx->ws); + ctx->depth_prev_va = bo->gpu_va; + if (bo->gpu_va) + sgx_bo_unbind(ctx->ws, bo); + ret = sgx_bo_bind_at(ctx->ws, bo, sgx_ctx_window[SGX_CTX_BO_DEPTH], + sgx_ctx_va[SGX_CTX_BO_DEPTH]); + /* The one it displaced goes back where it was, because a buffer that + * was a depth attachment is usually about to be sampled. */ + if (prev && + (!prev_va || sgx_bo_bind_at(ctx->ws, prev, + sgx_ctx_window[SGX_CTX_BO_DEPTH], + prev_va))) + sgx_bo_bind(ctx->ws, prev, sgx_ctx_window[SGX_CTX_BO_DEPTH]); + if (prev && SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: depth: displaced bo %p wanted 0x%llx, " + "now at 0x%llx\n", (void *)prev, + (unsigned long long)prev_va, + (unsigned long long)prev->gpu_va); + if (ret) { + ctx->depth_prev_va = 0; + return ret; + } + ctx->depth_bo = bo; + ctx->bo[SGX_CTX_BO_DEPTH] = *bo; + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: bind depth bo %p -> 0x%llx, %llu " + "byte(s) for %llu\n", (void *)bo, + (unsigned long long)bo->gpu_va, + (unsigned long long)bo->size, + (unsigned long long)need); + return 0; +} + +int sgx_release_depth(struct sgx_context *ctx) +{ + struct sgx_bo *prev; + uint64_t prev_va; + + if (!ctx || !ctx->depth_bo) + return 0; + prev = ctx->depth_bo; + prev_va = ctx->depth_prev_va; + if (prev->gpu_va == sgx_ctx_va[SGX_CTX_BO_DEPTH]) { + if (sgx_bo_is_busy(ctx->ws, prev)) + sgx_wait_idle(ctx->ws); + sgx_bo_unbind(ctx->ws, prev); + if (prev_va && + sgx_bo_bind_at(ctx->ws, prev, + sgx_ctx_window[SGX_CTX_BO_DEPTH], prev_va)) + sgx_bo_bind(ctx->ws, prev, + sgx_ctx_window[SGX_CTX_BO_DEPTH]); + } + ctx->bound[SGX_CTX_BO_DEPTH] = 0; + memset(&ctx->bo[SGX_CTX_BO_DEPTH], 0, + sizeof ctx->bo[SGX_CTX_BO_DEPTH]); + ctx->depth_bo = NULL; + ctx->depth_prev_va = 0; + /* The depth the frame had is not this object's, so nothing may be + * loaded back into the next pass from it. */ + ctx->depth_stored = 0; + return 0; +} + +int sgx_release_target(struct sgx_context *ctx) +{ + struct sgx_bo *prev; + uint64_t prev_va; + + if (!ctx) + return -EINVAL; + if (ctx->target_is_scanout || !ctx->target_bo) + return 0; + prev = ctx->target_bo; + prev_va = ctx->target_prev_va; + if (ctx->bound[SGX_CTX_BO_TARGET]) { + sgx_bo_free(ctx->ws, &ctx->bo[SGX_CTX_BO_TARGET]); + } else if (prev->gpu_va == sgx_ctx_va[SGX_CTX_BO_TARGET]) { + /* A deferred render may still be writing it, so it cannot be + * moved out from under the core - the same rule sgx_use_target() + * follows when it swaps one target for another. */ + if (sgx_bo_is_busy(ctx->ws, prev)) + sgx_wait_idle(ctx->ws); + sgx_bo_unbind(ctx->ws, prev); + /* Back where it was, because a buffer that was a render target + * is usually about to be sampled - and anywhere at all if it + * had no address to go back to. + * + * That second half was missing: with prev_va zero the test + * short-circuited and neither bind ran, so the buffer came out + * of here with no address while the caller still held it. The + * frame that named it next passed the address check further + * up - the winsys still believed the address it had - and the + * part faulted inside the target's own slot with nothing bound + * there. sgx_use_target() spells the same thing !prev_va || + * bind_at(), and this is now the same sentence. */ + if ((!prev_va || + sgx_bo_bind_at(ctx->ws, prev, + sgx_ctx_window[SGX_CTX_BO_TARGET], prev_va)) && + sgx_bo_bind(ctx->ws, prev, + sgx_ctx_window[SGX_CTX_BO_TARGET])) + fprintf(sgx_log(), "sgx: the released target kept no " + "address; a frame naming it will be refused\n"); + } + ctx->bound[SGX_CTX_BO_TARGET] = 0; + memset(&ctx->bo[SGX_CTX_BO_TARGET], 0, + sizeof ctx->bo[SGX_CTX_BO_TARGET]); + ctx->target_bo = NULL; + ctx->target_prev_va = 0; + return 0; +} + +int sgx_use_texture(struct sgx_context *ctx, unsigned unit, struct sgx_bo *bo) +{ + int ret; + struct sgx_bo *prev = NULL; + uint64_t prev_va = 0; + int alias; + + enum sgx_ctx_bo slot; + + if (!ctx || !bo) + return -EINVAL; + /* One buffer per unit the frame's relocations name an address for - + * the captured frame reserves two. + * + * A unit past those needs no fixed address at all: a record carries + * the texture's own base, which is what let a texture change cost a + * draw record rather than a whole submit. So it is bound wherever the + * texture window has room and noted, rather than refused - refusing + * left the view unbound and the draw was rejected whole. */ + if (unit >= SGX_MAX_TEX_UNITS) + return -ENOTSUP; + if (unit == 0) + slot = SGX_CTX_BO_TEX; + else if (unit == 1) + slot = SGX_CTX_BO_TEX2; + else { + if (!bo->gpu_va) { + ret = sgx_bo_bind(ctx->ws, bo, + sgx_ctx_window[SGX_CTX_BO_TEX]); + if (ret) + return ret; + } + ctx->tex_bo[unit] = bo; + ctx->tex_prev_va[unit] = bo->gpu_va; + return 0; + } + if (ctx->tex_bo[unit] == bo) + return 0; + /* The buffer arriving at this unit can be the one the frame renders + * into. sgx_use_target() has handled that from its side since a moved + * window came out as the fallback checkerboard - the incoming target + * can be a bound texture, and the unit follows it rather than the + * buffer being evicted - and this is the same case seen from the + * other end, which was never handled at all. + * + * Unhandled, it unbinds the buffer from the target's fixed address + * and rebinds it at this unit's. ctx->target_bo still points at it, + * but ctx->bo[SGX_CTX_BO_TARGET] is a copy by value and still holds + * the address it no longer has - so the next frame is built against + * an address nothing is bound at, and the part faults inside the + * target's own slot. The state tracker does not have to be asking + * for anything illegal for this to happen: it sets sampler views and + * the framebuffer in separate calls, this driver moves buffers on + * each of them, and a legal end state can pass through an + * intermediate one where the new texture is still the old target. + * + * So the target keeps its address and the unit reads it there. */ + alias = ctx->target_bo == bo; + if (ctx->bound[slot]) { + sgx_bo_free(ctx->ws, &ctx->bo[slot]); + ctx->bound[slot] = 0; + } else if (ctx->tex_bo[unit] && + ctx->tex_bo[unit]->gpu_va == sgx_ctx_va[slot]) { + /* The mirror of the same rule: a texture that has since become + * the render target is no longer this unit's to move. */ + sgx_bo_unbind(ctx->ws, ctx->tex_bo[unit]); + prev = ctx->tex_bo[unit]; + prev_va = ctx->tex_prev_va[unit]; + } + if (alias) { + /* The outgoing texture goes back first: this unit is not + * taking the fixed address, so nothing is waiting for it. */ + if (prev && + (!prev_va || sgx_bo_bind_at(ctx->ws, prev, + sgx_ctx_window[slot], prev_va))) + sgx_bo_bind(ctx->ws, prev, sgx_ctx_window[slot]); + ctx->tex_bo[unit] = bo; + ctx->bo[slot] = *bo; + ctx->tex_prev_va[unit] = 0; + if (ctx->view_bound & (1u << unit)) { + ctx->views[unit].gpu_va = (uint32_t)bo->gpu_va; + ctx_refresh_sampler_state(ctx); + } + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: tex%u is the render target; " + "the unit follows it to 0x%llx\n", unit, + (unsigned long long)bo->gpu_va); + return 0; + } + ctx->tex_prev_va[unit] = bo->gpu_va; + if (bo->gpu_va) + sgx_bo_unbind(ctx->ws, bo); + ret = sgx_bo_bind_at(ctx->ws, bo, sgx_ctx_window[slot], + sgx_ctx_va[slot]); + /* The outgoing texture gets an address back only once the incoming one + * holds the fixed address the frame names. Rehoming it first let + * "anywhere" take that address, which made this bind fail and left the + * unit unmapped - a render that faulted on the texture fetch. */ + if (prev && + (!prev_va || sgx_bo_bind_at(ctx->ws, prev, sgx_ctx_window[slot], + prev_va))) + sgx_bo_bind(ctx->ws, prev, sgx_ctx_window[slot]); + if (ret) { + /* The unit has to keep something mapped behind it. The frame + * names this slot's address in a relocation whether or not + * anything samples it, so a slot left empty - which is what + * this path did, having already freed the context's own + * buffer above - put a zero there, and the render read it: + * "MMU fault at 0x00000000, requestors: cache". Put the + * context's own texture back and drop the view, so the draw + * comes out untextured rather than stopping the core. */ + ctx->tex_bo[unit] = NULL; + ctx->view_bound &= ~(1u << unit); + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: tex%u would not bind at " + "0x%llx (%d); the unit keeps its own\n", unit, + (unsigned long long)sgx_ctx_va[slot], ret); + if (ctx_alloc(ctx, slot, sgx_ctx_size[slot])) + return ret; + return ret; + } + ctx->tex_bo[unit] = bo; + ctx->bo[slot] = *bo; + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: bind tex%u bo %p -> 0x%llx size %llu\n", + unit, (void *)bo, (unsigned long long)bo->gpu_va, + (unsigned long long)bo->size); + return 0; +} + +/* Give a unit's address back to the context's own texture. The frame samples + * one unit whatever the program does, so an address a caller's object has + * vacated has to have something mapped behind it again. */ +/* Whether a draw record in the frame being built names this buffer. + * + * A record captures the texture's address when the draw is made and the frame + * is submitted later, so the buffer has to outlive the frame. A client that + * drops a texture between drawing with it and the flush - which is ordinary + * for one that reuses its texture objects - left the record naming memory + * that had gone, and the render read it. */ +/* Adopt a caller's texture for a unit without moving it. The unit's fixed + * address is only needed when the frame relocates from it; a record that + * names the texture's own address needs nothing but this. */ +/* Forget a texture the frame was going to name, because it is being freed. */ +void sgx_frame_drop_tex(struct sgx_context *ctx, const struct sgx_bo *bo) +{ + unsigned i, n = 0; + + if (!ctx) + return; + for (i = 0; i < ctx->nframe_tex; i++) + if (ctx->frame_tex[i] != bo) + ctx->frame_tex[n++] = ctx->frame_tex[i]; + ctx->nframe_tex = n; +} + +/* Make a texture's most recent contents visible to the part. The kernel + * flushes an object's pages when it binds them, and a texture is bound once, + * when it is created - long before anything is uploaded into it. Moving it + * onto the unit's address was carrying that flush by accident; a texture left + * where it is needs it asked for. Rebinding at the address it already has + * costs a page-table write, not a frame. */ +int sgx_refresh_texture(struct sgx_context *ctx, struct sgx_bo *bo) +{ + uint64_t va; + + if (!ctx || !bo || !bo->gpu_va) + return -EINVAL; + va = bo->gpu_va; + sgx_bo_unbind(ctx->ws, bo); + return sgx_bo_bind_at(ctx->ws, bo, sgx_ctx_window[SGX_CTX_BO_TEX], va); +} + +int sgx_note_texture(struct sgx_context *ctx, unsigned unit, struct sgx_bo *bo) +{ + if (!ctx || unit >= SGX_MAX_TEX_UNITS) + return -EINVAL; + ctx->tex_bo[unit] = bo; + return 0; +} + +/* Note a texture this frame samples. Returns -ENOSPC when the submit has no + * room left for another, which is the caller's cue to end the frame. */ +int sgx_frame_add_tex(struct sgx_context *ctx, struct sgx_bo *bo) +{ + unsigned i; + + if (!ctx || !bo) + return 0; + for (i = 0; i < ctx->nframe_tex; i++) + if (ctx->frame_tex[i] == bo) + return 0; + if (ctx->nframe_tex >= SGX_FRAME_MAX_TEX) + return -ENOSPC; + ctx->frame_tex[ctx->nframe_tex++] = bo; + return 0; +} + +int sgx_context_frame_uses(const struct sgx_context *ctx, + const struct sgx_bo *bo) +{ + unsigned k; + + if (!ctx || !bo || !bo->gpu_va || !ctx->draws) + return 0; + /* Both units. Checking only the first let a texture sampled solely by + * the second be freed while a record still carried its address, so the + * frame sampled memory that had gone back to the allocator. */ + for (k = 0; k < ctx->nrange; k++) { + if (ctx->range[k].tex[0].bound && + ctx->range[k].tex[0].va == (uint32_t)bo->gpu_va) + return 1; + if (ctx->range[k].tex[1].bound && + ctx->range[k].tex[1].va == (uint32_t)bo->gpu_va) + return 1; + } + return 0; +} + +int sgx_release_texture(struct sgx_context *ctx, unsigned unit) +{ + enum sgx_ctx_bo slot; + int ret; + + if (!ctx || unit >= SGX_MAX_TEX_UNITS) + return -EINVAL; + slot = unit ? SGX_CTX_BO_TEX2 : SGX_CTX_BO_TEX; + if (!ctx->tex_bo[unit]) + return 0; /* already the context's own */ + /* The caller's buffer is going, so stop naming it whatever else is + * true. A texture that was never moved onto the unit's address leaves + * the context's own buffer bound there, and returning early on that + * kept a pointer to memory the caller had freed - which the next + * frame described, and nothing drew from then on. */ + if (ctx->bound[slot]) { + ctx->tex_bo[unit] = NULL; + ctx->tex_prev_va[unit] = 0; + sgx_set_sampler_view(ctx, unit, NULL); + return ctx->cookie_valid ? ctx_build_frame(ctx) : 0; + } + if (ctx->tex_bo[unit]->gpu_va == sgx_ctx_va[slot]) + sgx_bo_unbind(ctx->ws, ctx->tex_bo[unit]); + ctx->tex_bo[unit] = NULL; + ctx->tex_prev_va[unit] = 0; + /* The descriptor is written from the bound view, so the view has to go + * with the buffer - otherwise the frame keeps describing the object + * that left. Sampling what is now the render target is undefined in + * any case. */ + sgx_set_sampler_view(ctx, unit, NULL); + ret = ctx_alloc(ctx, slot, sgx_ctx_size[slot]); + if (ret) + return ret; + /* The heap still describes the object that left - its width, height + * and format - so the frame kept sampling those dimensions out of the + * context's much smaller one and ran off the end of it. Rebuilding the + * frame puts the descriptor back with the buffer. */ + return ctx->cookie_valid ? ctx_build_frame(ctx) : 0; +} + +int sgx_set_clear_enable(struct sgx_context *ctx, int on) +{ + if (!ctx) + return -EINVAL; + on = !!on; + /* Turning the clear off ends the frame's clear entirely, depth + * included - every caller that does it means "the next frame continues + * this picture". Turning it on clears both, as it always has. */ + if (ctx->clear_on == on && (on || !ctx->zclear_on)) + return 0; + ctx->clear_on = on; + if (!on) + ctx->zclear_on = 0; + /* The clearing draws are part of the frame's records and relocations, + * so the frame has to be built again without them. */ + return ctx->cookie_valid ? ctx_build_frame(ctx) : 0; +} + +int sgx_set_zclear_enable(struct sgx_context *ctx, int on) +{ + if (!ctx) + return -EINVAL; + on = !!on; + if (ctx->zclear_on == on) + return 0; + ctx->zclear_on = on; + return ctx->cookie_valid ? ctx_build_frame(ctx) : 0; +} + +void sgx_request_clear(struct sgx_context *ctx) +{ + if (ctx) + ctx->clear_pending = 1; +} + +/* Copy the caller's fragment uniforms into the heap block the secondary PDS + * program DMAs into the sa bank. Bounded by what the program actually declares + * and by what the caller supplied, exactly as ctx_sec_pds() does when it + * builds the block in the first place. */ +/* A shader-issued sample names three consecutive secondary attributes holding + * the unit's state - control, format, address - and the backend reserves a + * quad per sampler behind the uniforms for exactly that. Nothing wrote them, + * so the sampler read whatever the bank last held. The address is the bound + * one: the frame's relocation puts a pre-iterated unit's address in the PDS + * data segment, which a program sampling for itself never reads. */ +/* Into a given block, laid out for a given program: which program decides + * where the slots sit, so a record running its own reads its own offsets. */ +/* Which state block a unit's sample reads first, and how many planes the + * program was built to sample there. The backend reports the map for the + * units the program names; a unit past them - a bound view the program + * never reads - gets a block of its own behind the last, so that writing + * its descriptor lands nowhere the code reads from. */ +static unsigned fs_unit_slot(const struct sgx_shader *fs, unsigned u, + unsigned *nchunks) +{ + /* The slot, as the register the shader's rule names less the base. */ + return (sgx_shader_smp_reg(fs, u, 0, 0, nchunks) - + sgx_shader_smp_reg(fs, 0, 0, 0, NULL)) / SGX_SMP_SLOT + + (fs && fs->smp_slot && fs->nsamp ? fs->smp_slot[0] : 0u); +} + +/* How many sampler slots the backend laid out for a program: every unit it + * samples, and every view the caller bound, whichever is more. */ +static unsigned ctx_smp_units(const struct sgx_context *ctx, + const struct sgx_shader *fs) +{ + unsigned u, nu, ntex = 0; + + if (!ctx || !fs) + return 0; + nu = ctx->nviews; + for (u = 0; u < sizeof fs->tex_units / sizeof fs->tex_units[0]; u++) + ntex += fs->tex_units[u]; + if (ntex > nu) + nu = ntex; + /* A program that samples for itself needs a descriptor whatever the + * counts say - tex_units is keyed by the input the coordinate comes + * from, and one reaching the sample through a temporary leaves it + * empty. */ + if (!nu && !fs->tex_preiterated) + nu = 1; + /* Never hand back more units than the program reserved registers for. + * The literal pool starts right after the descriptors, so a unit the + * shader never samples - a fallback with the default address - is + * written straight over the constants. In blocks, since a unit whose + * texel is stored in planes takes one block a plane. */ + if (fs->pool_base > fs->smp_base) { + unsigned room = (fs->pool_base - fs->smp_base) / SGX_SMP_SLOT; + unsigned u; + + for (u = 0; u < nu; u++) { + unsigned n; + + if (fs_unit_slot(fs, u, &n) + n > room) { + nu = u; + break; + } + } + } + if (nu > SGX_MAX_SAMPLERS) + nu = SGX_MAX_SAMPLERS; + return nu; +} + +/* Where the backend actually put a program's sampler descriptors. + * + * They sit between the uniforms and the literal pool - the backend lays the + * bank out as uniforms, then a quad per sampler, then the pool - so the pool's + * base less a quad per sampler is where the first one is. nuniform is not that + * number: it under-reports, four of twelve programs in a twm trace say zero + * while carrying a colour, and a descriptor written at four times it lands + * inside the uniforms, where the next constant upload overwrites it. The + * render then sampled at whatever float it found - 0x3f800000, 0x41401000 - + * with the texture cache as the requestor. */ +/* Where the code reads sampler 0's first state block. One rule, in + * sgx_shader.c, so this and the compiled SMP cannot disagree - they were + * four registers apart on a program whose base is sa0, and the part read + * the shifted words as an address and faulted at 0 and at the control + * word itself. */ +static unsigned fs_smp_base(const struct sgx_shader *fs, unsigned nu) +{ + return sgx_shader_smp_reg(fs, 0, 0, nu, NULL) - + SGX_SMP_SLOT * (fs && fs->smp_slot && fs->nsamp ? + fs->smp_slot[0] : 0u); +} + +/* The record's own textures when it has them, the bound ones otherwise. The + * iterated path already describes each record's textures separately, and this + * one did not: it read ctx->views[] at flush, so every draw in a frame sampled + * with the dimensions of whichever texture happened to be bound last. There is + * no pitch in a descriptor - the address of a texel is computed from the + * declared width - so a wrong width does not merely pick the wrong picture, it + * repeats it. That is ioquake3's HUD and font drawn as grids of their whole + * atlas over a correct 3D scene. */ +static void ctx_write_sampler_slots(struct sgx_context *ctx, + const struct sgx_shader *fs, + const struct sgx_ctx_range *r, + uint32_t *blk, uint32_t blk_off) +{ + unsigned u, nu; + + if (!ctx || !fs || !blk) + return; + /* Every slot the backend laid out, not only the ones the caller + * bound: a program that samples a unit the caller left unbound read + * the slot as zero and faulted at address zero. */ + nu = ctx_smp_units(ctx, fs); + for (u = 0; u < nu; u++) { + unsigned nch, base = sgx_shader_smp_reg(fs, u, 0, nu, &nch); + const struct sgx_sampler_view *v = &ctx->views[u]; + const struct sgx_sampler_state *st = &ctx->samplers[u]; + struct sgx_sampler_view dv; + struct sgx_sampler_state dst; + uint32_t w0 = 0, w1 = 0, vch = 1, vcs = 0; + + if (blk_off + (base + SGX_SMP_SLOT * nch) * 4u > + ctx->bo[SGX_CTX_BO_HEAP].size) + return; + /* Skipping an undescribed unit left the slot holding zero and + * the sample read address zero, which stopped the core with + * the texture cache as the requestor. Describe what is mapped + * instead, exactly as the iterated path does. */ + if (u >= ctx->nviews || !(ctx->view_bound & (1u << u)) || + !v->gpu_va) { + uint64_t va = ctx->bo[SGX_CTX_BO_TEX].gpu_va; + + /* One entry per texture unit, while the loop runs + * over every sampler the backend laid out - a third + * sampler read past the array and dereferenced a + * neighbouring address as a pointer. */ + if (!va && u < SGX_MAX_TEX_UNITS && ctx->tex_bo[u]) + va = ctx->tex_bo[u]->gpu_va; + if (!va) + va = ctx->bo[SGX_CTX_BO_HEAP].gpu_va; + if (!va) + continue; + memset(&dv, 0, sizeof dv); + memset(&dst, 0, sizeof dst); + dv.gpu_va = (uint32_t)va; + dv.width = 32u; + dv.height = 1u; + dv.stride = 32u * 4u; + dv.format = SGX_FMT_A8R8G8B8; + dv.nlevels = 1; + v = &dv; + st = &dst; + } + { + uint32_t rva = 0, k; + + if (r && u < SGX_MAX_TEX_UNITS && r->tex[u].bound && + r->tex[u].va) { + w0 = r->tex[u].w0; w1 = r->tex[u].w1; + rva = r->tex[u].va; + vch = r->tex[u].nchunks; + vcs = r->tex[u].chunk_size; + } else if (sgx_sampler_words(v, st, &w0, &w1) != + SGX_SAMPLER_OK) { + continue; + } else { + vch = v->nchunks; + vcs = v->chunk_size; + } + if (!vch) + vch = 1; + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), + "sgx: smp unit %u: %s va 0x%x (nviews %u, bound 0x%x, nu %u, base %u)\n", + u, rva ? "record" : + v == &dv ? "FALLBACK" : "caller", + rva ? rva : (uint32_t)v->gpu_va, + ctx->nviews, ctx->view_bound, nu, base); + /* One block a plane, the planes chunk_size apart - + * the vendor's aui32StateWord2[i] (texmgmt.c:1231) - + * from the map's base for a border mode, the texels + * else. A view with fewer planes than the program + * samples repeats its last, and a program built for + * fewer than the view holds reads only those: the + * class key rebuilds the program before either + * matters. */ + for (k = 0; k < nch; k++) { + uint32_t va = rva ? rva : + sgx_sampler_view_base(v, st); + uint32_t p = k < vch ? k : vch - 1u; + + blk[base + SGX_SMP_SLOT * k + 0] = w0; + blk[base + SGX_SMP_SLOT * k + 1] = w1; + blk[base + SGX_SMP_SLOT * k + 2] = va + p * vcs; + } + } + } +} + +static void ctx_refresh_sampler_state(struct sgx_context *ctx) +{ + uint32_t *heap; + + if (!ctx || !ctx->fs || ctx->fs->tex_preiterated) + return; + heap = ctx->bo[SGX_CTX_BO_HEAP].map; + if (!heap) + return; + /* The sampler descriptors are in the block the render's secondary PDS + * program DMAs, so this cannot be written under a render that is still + * sampling. After the early returns above, so a bind that writes + * nothing costs nothing. */ + sgx_wait_idle(ctx->ws); + ctx_write_sampler_slots(ctx, ctx->fs, NULL, heap + SGX_FS_CONST_OFF / 4, + SGX_FS_CONST_OFF); +} + +/* Mesa lays a constant buffer out as one vec4 per uniform; the secondary + * attribute bank holds each uniform at the width the program reads it through, + * which is what keeps a shader with several scalar uniforms inside a + * thirty-two register bank. So the upload gathers rather than copies, into the + * layout the backend reported. Returns the registers written. */ +static unsigned ctx_pack_uniforms(const struct sgx_shader *fs, + const uint32_t *src, unsigned src_dwords, + uint32_t *dst, unsigned dst_dwords) +{ + unsigned i, top = 0; + + if (!fs || !src || !dst) + return 0; + for (i = 0; i < fs->nuniform; i++) { + unsigned b = fs->uni_base[i], k; + unsigned w = fs->uni_width[i] ? fs->uni_width[i] : 4u; + + for (k = 0; k < w; k++) { + if (4u * i + k >= src_dwords || b + k >= dst_dwords) + break; + dst[b + k] = src[4u * i + k]; + } + if (b + w > top) + top = b + w; + } + return top; +} + +static void ctx_refresh_fs_constants(struct sgx_context *ctx) +{ + uint32_t *heap; + unsigned nuni; + + if (!ctx || !ctx->fs || !ctx->fs_const_dwords) + return; + heap = ctx->bo[SGX_CTX_BO_HEAP].map; + if (!heap) + return; + if (SGX_FS_CONST_OFF + ctx->fs_const_dwords * 4u > + ctx->bo[SGX_CTX_BO_HEAP].size) + return; + /* Same block, same reason as ctx_refresh_sampler_state(). */ + sgx_wait_idle(ctx->ws); + /* What the caller bound, not what the program declared: the secondary + * program DMAs pool_base + pool_dwords whatever this copies, so a + * program reporting no uniforms read the block's previous contents. + * nuniform under-reports - four of twelve programs in a twm trace say + * zero while carrying a colour. pool_base bounds it, the block being + * uniforms, then sampler slots, then the literal pool. */ + nuni = ctx->fs_const_dwords; + if (nuni > ctx->fs->pool_base) + nuni = ctx->fs->pool_base; + /* pool_base spans the sampler slots as well, so bounding by it alone + * let a caller's constants run over the descriptor a self-sampling + * program reads. The sample then issued against constant data and + * never completed, and the tiler waited for a texel that never came: + * ioquake3's first textured frame stalled the core for it. The + * program's own uniform count is what the backend laid the bank out + * with, so that is the bound here. */ + /* Also when the program declares no uniforms at all: the clamp used to + * be skipped for nuniform == 0, so pool_base alone bounded the copy + * and the caller's constants landed on the sampler descriptor that + * follows. The sample then read address zero and the render stopped + * with the texture cache as the requestor. */ + if (!ctx->fs->tex_preiterated) { + unsigned lim = fs_smp_base(ctx->fs, ctx_smp_units(ctx, ctx->fs)); + + if (nuni > lim) + nuni = lim; + } + if (nuni) + ctx_pack_uniforms(ctx->fs, ctx->fs_const, + ctx->fs_const_dwords, + heap + SGX_FS_CONST_OFF / 4, nuni); + /* Constants and sampler slots share the block, and the caller sets + * them in either order; the descriptor has to survive both. */ + ctx_refresh_sampler_state(ctx); +} + +int sgx_set_fs_constants(struct sgx_context *ctx, const uint32_t *v, + unsigned n) +{ + if (!ctx) + return -EINVAL; + if (n > SGX_MAX_FS_CONST_DW) { + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: fragment uniforms: %u dwords " + "truncated to %u\n", n, SGX_MAX_FS_CONST_DW); + n = SGX_MAX_FS_CONST_DW; + } + if (n && v) + memcpy(ctx->fs_const, v, n * sizeof(*v)); + ctx->fs_const_dwords = (n && v) ? n : 0; + /* And put them where the secondary PDS program reads them. The block + * was only ever filled while the frame was being built - which happens + * on a framebuffer or clear-colour change, not on a uniform change - + * so every later glUniform call was kept in this struct and never + * reached the hardware. The shader went on using whatever the frame + * was built with, which is why a sequence of draws all came out the + * colour of the first one. */ + ctx_refresh_fs_constants(ctx); + return 0; +} + +int sgx_clear_pending(const struct sgx_context *ctx) +{ + return ctx && ctx->clear_pending; +} + +int sgx_set_clear_color(struct sgx_context *ctx, uint32_t packed) +{ + if (!ctx) + return -EINVAL; + /* All twenty-four bits are carried now: the immediate is spread over + * three fields of the instruction and xpsb_gen_usse() writes all of + * them, where it used to write only the low twenty-one. */ + packed &= 0x00ffffffu; + { + const char *f = SGX_ENVS("SGX_FORCE_CLEAR"); + + if (f) + packed = (uint32_t)strtoul(f, NULL, 0) & 0x00ffffffu; + } + if (ctx->clear_color == packed) + return 0; + ctx->clear_color = packed; + /* The colour is an immediate inside the frame's USSE programs, so the + * frame has to be built again to carry it. */ + return ctx->cookie_valid ? ctx_build_frame(ctx) : 0; +} + +int sgx_set_clear_depth(struct sgx_context *ctx, float depth) +{ + union { float f; uint32_t u; } v; + + if (!ctx) + return -EINVAL; + v.f = depth < 0.0f ? 0.0f : depth > 1.0f ? 1.0f : depth; + if (ctx->clear_depth == v.u) + return 0; + ctx->clear_depth = v.u; + /* In the clearing records' vertices, so the frame is rebuilt. */ + return ctx->cookie_valid ? ctx_build_frame(ctx) : 0; +} + +int sgx_set_clear_stencil(struct sgx_context *ctx, unsigned stencil) +{ + if (!ctx) + return -EINVAL; + /* A register in the submit, so nothing to rebuild. */ + ctx->clear_stencil = stencil & 0xffu; + return 0; +} + +int sgx_use_scanout(struct sgx_context *ctx) +{ + struct sgx_scanout so; + int ret; + + if (!ctx) + return -EINVAL; + ret = sgx_get_scanout(ctx->ws, &so); + if (ret) + return ret; + /* The kernel maps it where an allocated target would have gone, so the + * frame's relocations need no adjusting. */ + if (so.gpu_va != sgx_ctx_va[SGX_CTX_BO_TARGET]) + return -EINVAL; + /* The relocations source the target's address from the object, so it + * has to carry the scanout's even though nothing was allocated for it + * and nothing is bound: the kernel already mapped it. */ + ctx->bo[SGX_CTX_BO_TARGET].gpu_va = so.gpu_va; + ctx->bo[SGX_CTX_BO_TARGET].size = so.size; + ctx->target_is_scanout = 1; + ctx->scanout_pitch = so.pitch; + ctx->scanout_w = so.width; + ctx->scanout_h = so.height; + return 0; +} + +int sgx_set_max_prims(struct sgx_context *ctx, unsigned prims) +{ + if (!ctx || !prims) + return -EINVAL; + ctx->max_prims = prims; + return 0; +} + +int sgx_set_hwtcl(struct sgx_context *ctx, int on) +{ + if (!ctx) + return -EINVAL; + on = !!on; + if (ctx->hwtcl == on) + return 0; + ctx->hwtcl = on; + ctx->isp_dirty = 1; /* word B's bias follows the transform */ + ctx->vtx_floats = on ? ctx_hwtcl_record_floats(ctx, ctx->vs_const_dwords) : + SGX_VTX_FLOATS; + /* The record's width is built into the frame - the vertex DMA's + * control word and the fetch program's stride both carry it - so + * changing it after a framebuffer is bound means building the frame + * again. Leaving the old one is what drew nothing at all. */ + if (ctx->cookie_valid) + return sgx_set_framebuffer(ctx, &ctx->fb); + return 0; +} + +int sgx_set_vs_fixed(struct sgx_context *ctx, int on) +{ + if (!ctx) + return -EINVAL; + ctx->vs_fixed = !!on; + return 0; +} + +int sgx_set_vs_constants(struct sgx_context *ctx, const float *v, unsigned n) +{ + if (!ctx || !v || !n) + return -EINVAL; + if (n > SGX_HWTCL_MAX_UNI_LIMM) + return -EINVAL; + if (!sgx_hwtcl_limm_uniforms() && n > SGX_HWTCL_MAX_UNI) + return -EINVAL; + memcpy(ctx->vs_const, v, n * sizeof *v); + ctx->vs_const_dwords = n; + if (ctx->hwtcl) + ctx->vtx_floats = ctx_hwtcl_record_floats(ctx, n); + return 0; +} + +/* The set index for what is bound now, adding it if the frame has not seen it. + * Returns SGX_HWTCL_MAX_UNI_SETS when the frame is full of them. */ +/* Room for one more set in either store, growing it if there is not. The two + * differ only in how wide an entry is, so the sizes come in as arguments. */ +static int ctx_uni_grow(void **set, unsigned **dwords, unsigned *cap, + unsigned n, unsigned entry, unsigned max) +{ + unsigned nc = *cap ? *cap * 2u : SGX_UNI_SETS_INITIAL; + void *p, *q; + + if (n < *cap) + return 0; + if (n >= max) + return -ENOSPC; + if (nc > max) + nc = max; + p = realloc(*set, (size_t)nc * entry); + if (!p) + return -ENOMEM; + *set = p; + q = realloc(*dwords, (size_t)nc * sizeof **dwords); + if (!q) + return -ENOMEM; + *dwords = q; + *cap = nc; + return 0; +} + +static unsigned ctx_uni_set(struct sgx_context *ctx) +{ + unsigned i; + + for (i = 0; i < ctx->nuni_set; i++) + if (ctx->uni_set_dwords[i] == ctx->vs_const_dwords && + !memcmp(ctx->uni_set[i], ctx->vs_const, + ctx->vs_const_dwords * sizeof ctx->vs_const[0])) + return i; + if (ctx_uni_grow((void **)&ctx->uni_set, &ctx->uni_set_dwords, + &ctx->uni_set_cap, ctx->nuni_set, + sizeof *ctx->uni_set, SGX_HWTCL_MAX_UNI_SETS)) + return SGX_HWTCL_MAX_UNI_SETS; + memcpy(ctx->uni_set[ctx->nuni_set], ctx->vs_const, + ctx->vs_const_dwords * sizeof ctx->vs_const[0]); + ctx->uni_set_dwords[ctx->nuni_set] = ctx->vs_const_dwords; + return ctx->nuni_set++; +} + +/* Which fragment-uniform set the draw about to open a record would use, + * adding it if it is new. Mirrors ctx_uni_set() on the vertex side: records + * that share their uniforms share a set, so a frame that never changes them + * carries exactly one. Returns SGX_FS_MAX_UNI_SETS when the frame is full, + * which the caller turns into a refusal rather than the wrong colours. */ +static unsigned ctx_fs_uni_set(struct sgx_context *ctx) +{ + unsigned i; + + if (!ctx->fs_const_dwords) + return 0; + /* The set this draw wants is nearly always the one the draw before + * it wanted, and the scan below is every set against a memcmp each - + * the hottest loop in sgx_draw() on a scene with many of them. */ + i = ctx->fs_uni_set_last; + if (i < ctx->nfs_uni_set && + ctx->fs_uni_set_dwords[i] == ctx->fs_const_dwords && + !memcmp(ctx->fs_uni_set[i], ctx->fs_const, + ctx->fs_const_dwords * sizeof ctx->fs_const[0])) + return i; + for (i = 0; i < ctx->nfs_uni_set; i++) + if (ctx->fs_uni_set_dwords[i] == ctx->fs_const_dwords && + !memcmp(ctx->fs_uni_set[i], ctx->fs_const, + ctx->fs_const_dwords * sizeof ctx->fs_const[0])) { + ctx->fs_uni_set_last = i; + return i; + } + if (ctx_uni_grow((void **)&ctx->fs_uni_set, &ctx->fs_uni_set_dwords, + &ctx->fs_uni_set_cap, ctx->nfs_uni_set, + sizeof *ctx->fs_uni_set, SGX_FS_MAX_UNI_SETS)) + return SGX_FS_MAX_UNI_SETS; + memcpy(ctx->fs_uni_set[ctx->nfs_uni_set], ctx->fs_const, + ctx->fs_const_dwords * sizeof ctx->fs_const[0]); + ctx->fs_uni_set_dwords[ctx->nfs_uni_set] = ctx->fs_const_dwords; + ctx->fs_uni_set_last = ctx->nfs_uni_set; + return ctx->nfs_uni_set++; +} + +/* Whether a draw with the uniforms bound now can join the open record. */ +int sgx_vs_constants_changed(const struct sgx_context *ctx) +{ + unsigned k; + + if (!ctx || !ctx->hwtcl || !ctx->nrange) + return 0; + k = ctx->range[ctx->nrange - 1].uni; + if (k >= ctx->nuni_set) + return 1; + return ctx->uni_set_dwords[k] != ctx->vs_const_dwords || + memcmp(ctx->uni_set[k], ctx->vs_const, + ctx->vs_const_dwords * sizeof ctx->vs_const[0]) != 0; +} + +int sgx_set_vs_matrix(struct sgx_context *ctx, const float *m) +{ + return sgx_set_vs_constants(ctx, m, SGX_HWTCL_MAT_N); +} + +int sgx_set_vertex_buffers(struct sgx_context *ctx, + const struct sgx_vertex_buffer *vb, unsigned n) +{ + if (!ctx || n > SGX_MAX_VERTEX_ELEMENTS) + return -EINVAL; + if (n && !vb) + return -EINVAL; + memset(ctx->vb, 0, sizeof ctx->vb); + if (n) + memcpy(ctx->vb, vb, n * sizeof *vb); + ctx->nvb = n; + return 0; +} + +int sgx_set_vertex_elements(struct sgx_context *ctx, + const struct sgx_vertex_element *ve, unsigned n) +{ + unsigned i; + + if (!ctx || n > SGX_MAX_VERTEX_ELEMENTS) + return -EINVAL; + if (n && !ve) + return -EINVAL; + for (i = 0; i < n; i++) { + if (!ve[i].ncomp || ve[i].ncomp > 4) + return -EINVAL; + if (ve[i].vb_index >= SGX_MAX_VERTEX_ELEMENTS) + return -EINVAL; + /* the record is 11 floats and an element may not run off it */ + if (ve[i].slot + ve[i].ncomp > ctx->vtx_floats) + return -EINVAL; + } + memset(ctx->ve, 0, sizeof ctx->ve); + if (n) + memcpy(ctx->ve, ve, n * sizeof *ve); + ctx->nve = n; + return 0; +} + +/* Convert the caller's vertices into the hardware's record and put them where + * the draw record already points, exactly as sgxtri's frame_set_mesh() does. + * Positions are taken as they arrive: the frame's vertex program is a + * pass-through, so they are screen space, not clip space. */ +/* Room for one more draw record, grown as needed. Returns non-zero when there + * is none - which is a refusal, not a silent drop: the vertices are already + * written by the time this is asked. */ +static int ctx_range_room(struct sgx_context *ctx) +{ + unsigned want; + void *p; + + if (ctx->nrange < ctx->range_cap) + return 0; + want = ctx->range_cap ? ctx->range_cap * 2u : SGX_DRAWS_INITIAL; + if (want > SGX_MAX_DRAWS) + want = SGX_MAX_DRAWS; + if (want <= ctx->nrange) + return -ENOSPC; + p = realloc(ctx->range, (size_t)want * sizeof *ctx->range); + if (!p) + return -ENOMEM; + ctx->range = p; + ctx->range_cap = want; + return 0; +} + +/* Why a frame ended early, tallied per reason. A split is not free: the next + * pass starts against a far background with none of this one's depth, so + * anything drawn after it is no longer occluded by anything drawn before. + * SGX_SPLIT_STATS=1 reports the tally every 256 splits. */ +static struct { const char *why; unsigned n; } sgx_split_reason[16]; +static unsigned sgx_split_nreason, sgx_split_total; + +static int ctx_split(const char *why) +{ + unsigned i; + + for (i = 0; i < sgx_split_nreason; i++) + if (sgx_split_reason[i].why == why) + break; + if (i == sgx_split_nreason && i < 16) { + sgx_split_reason[i].why = why; + sgx_split_reason[i].n = 0; + sgx_split_nreason++; + } + if (i < 16) + sgx_split_reason[i].n++; + sgx_perf_note(why); + if ((++sgx_split_total == 1u || !(sgx_split_total % 32u)) && + SGX_ENV("SGX_SPLIT_STATS")) { + fprintf(sgx_log(), "sgx: %u frame split(s):", sgx_split_total); + for (i = 0; i < sgx_split_nreason; i++) + fprintf(sgx_log(), " %u x %s;", sgx_split_reason[i].n, + sgx_split_reason[i].why); + fputc('\n', stderr); + } + return -ENOSPC; +} + +static int ctx_upload_vertices(struct sgx_context *ctx, unsigned count) +{ + char *b = ctx->bo[SGX_CTX_BO_VTX].map; + float bx0 = 1e30f, by0 = 1e30f, bx1 = -1e30f, by1 = -1e30f; + const int trace = SGX_ENV("SGX_DMG_TRACE"); + float *out; + uint16_t *idx; + unsigned k, e, base, u, max_vtx, nidx; + struct { + const unsigned char *p; + unsigned stride, ncomp, slot; + int half, is_pos, rhw; + } src[SGX_MAX_VERTEX_ELEMENTS]; + size_t first_v = 0; + + if (!ctx->nve || !ctx->nvb) + return 0; /* keep the built-in triangle */ + if (!b) + return -EINVAL; + /* Against this draw's own record width, not the widest the frame could + * describe: at one coordinate set the record is twelve floats and a + * draw of 43044 fits, while the widest-record bound refused it and no + * number of splits could ever make a single draw smaller. */ + max_vtx = ctx->vtx_floats ? XPSB_MAX_VTX_AT(ctx->vtx_floats) + : XPSB_MAX_VTX; + if (count > max_vtx) { + sgx_ctx_dbg("upload: %u vertices at %u floats, past the %u a " + "frame holds\n", count, ctx->vtx_floats, max_vtx); + return ctx_split("vertices past the frame's maximum"); + } + /* Vertices and indices are separate counts once the caller supplies + * its own: a de-duplicated draw uploads fewer records than it names, + * and the index array is sized by the names. */ + nidx = ctx->draw_idx ? ctx->draw_nidx : count; + if ((size_t)XPSB_IDX_OFF + + ((size_t)ctx->vtx_uploaded + nidx) * 2 > ctx->bo[SGX_CTX_BO_VTX].size) + return ctx_split("the index array does not fit the buffer"); + + /* Draws accumulate into one frame: the state tracker issues several + * per frame and each is appended, because a flush is a whole frame - + * clear included - so one per draw would clear away the last. */ + base = ctx->vtx_uploaded; + if (base + nidx > max_vtx) { + sgx_ctx_dbg("upload: %u + %u vertices, past the %u a frame " + "holds\n", base, count, max_vtx); + return ctx_split("vertices past the frame's maximum"); + } + /* XPSB_MAX_VTX is derived from an eleven-float record, but the write + * below strides by vtx_floats - fourteen under hardware transform, and + * more again with a second coordinate set in the record. Bounding one + * against the other let the vertex array run into the index array the + * tiler reads, which submits a malformed frame rather than failing. */ + /* Placed by float offset and aligned so that the record's own first + * vertex number times its own stride is exactly that offset - which is + * what lets the fetch reach it with the record's stride. */ + { + size_t cur = ctx->vtx_float_cursor; + size_t s = ctx->vtx_floats; + + cur = (cur + s - 1) / s * s; + first_v = cur / s; + if (first_v + count > 0xffffu) { + sgx_ctx_dbg("upload: vertex %zu past what an index " + "holds\n", first_v + count); + return ctx_split("the vertex index does not fit"); + } + if ((size_t)XPSB_VTX_OFF + (cur + (size_t)count * s) * 4 > + (size_t)XPSB_IDX_OFF) { + sgx_ctx_dbg("upload: %u vertices at %u floats runs " + "into the index array\n", count, + ctx->vtx_floats); + return ctx_split("the vertex buffer is full"); + } + out = (float *)(b + XPSB_VTX_OFF) + cur; + ctx->vtx_float_cursor = cur + (size_t)count * s; + } + idx = (uint16_t *)(b + XPSB_IDX_OFF) + base; + + /* The loop below runs for every vertex of every draw, so what only + * depends on the element is resolved once here: the source base, the + * bounds at the last vertex, and whether the components can be taken + * in one copy instead of four bytes at a time. */ + for (e = 0; e < ctx->nve; e++) { + const struct sgx_vertex_element *v = &ctx->ve[e]; + const struct sgx_vertex_buffer *vb = &ctx->vb[v->vb_index]; + unsigned stride = v->stride ? v->stride : vb->stride; + uint64_t at; + + if (!vb->data || !stride) + return -EINVAL; + at = (uint64_t)vb->offset + (uint64_t)(count - 1) * stride + + v->src_offset; + if (at + (uint64_t)v->ncomp * sgx_vertex_comp_size(v) > + vb->size) + return -EINVAL; + src[e].p = (const unsigned char *)vb->data + vb->offset + + v->src_offset; + src[e].stride = stride; + /* The staged row is the record's width exactly, so an element + * that reaches past it is clamped rather than written into + * whatever follows. */ + src[e].ncomp = v->slot >= ctx->vtx_floats ? 0 : + (v->slot + v->ncomp > ctx->vtx_floats ? + ctx->vtx_floats - v->slot : v->ncomp); + src[e].slot = v->slot >= ctx->vtx_floats ? 0 : v->slot; + src[e].half = v->half; + src[e].is_pos = v->slot == 0; + src[e].rhw = v->slot == 0 && v->rhw && at + 16 <= vb->size && + !SGX_ENV("SGX_NO_PERSP"); + } + + /* The vertex buffer is write-combined, and the loop below writes the + * defaults and then overwrites part of them with the attributes: read + * back and scattered, which write combining cannot merge. Each record + * is assembled in ordinary memory and goes out in one run. */ + if (ctx->vtx_floats > ctx->vtx_row_cap) { + float *n = realloc(ctx->vtx_row, + ctx->vtx_floats * sizeof *n); + + if (!n) + return -ENOMEM; + ctx->vtx_row = n; + ctx->vtx_row_cap = ctx->vtx_floats; + } + for (k = 0; k < count; k++) { + float *o = ctx->vtx_row; + float oow; + + /* what the caller does not supply, the record still needs */ + o[0] = o[1] = o[2] = 0.0f; o[3] = 1.0f; + o[4] = o[5] = o[6] = o[7] = 1.0f; + o[8] = o[9] = 0.0f; o[10] = 1.0f; + /* the wide record's second coordinate set */ + for (e = 11; e < ctx->vtx_floats; e++) + o[e] = 0.0f; + + for (e = 0; e < ctx->nve; e++) { + const unsigned char *p = src[e].p + + (size_t)k * src[e].stride; + float *d = &o[src[e].slot]; + + if (!src[e].half) + memcpy(d, p, (size_t)src[e].ncomp * 4); + else { + unsigned i; + + for (i = 0; i < src[e].ncomp; i++) { + uint16_t h; + + memcpy(&h, p + i * 2, 2); + d[i] = sgx_half_to_float(h); + } + } + /* The software path uploads window coordinates, so + * the damage origin comes straight off them. */ + if (src[e].is_pos && ctx->dmg_on > 0 && !ctx->hwtcl) { + o[0] -= (float)ctx->dmg_x0; + o[1] -= (float)ctx->dmg_y0; + } + if (src[e].is_pos) { + if (o[0] < bx0) bx0 = o[0]; + if (o[1] < by0) by0 = o[1]; + if (o[0] > bx1) bx1 = o[0]; + if (o[1] > by1) by1 = o[1]; + } + /* The record's fourth position float is the RHW plane: + * the part builds every iteration plane from it - + * texture coordinates through the TAG's projection, + * colours through the use-issue's perspective bit - + * and G10's bit 12 (the DDK's WPRESENT) has declared + * it present in every captured frame. The draw module + * leaves exactly this value, 1/w, in the position it + * emits, which is also what the vendor's own vertex + * emit writes there (Mesa vf_generic.c, + * insert_4f_viewport_4: out[3] = in[3]). With 1.0 + * there instead every plane is affine, which was + * ioquake3's textures swimming across each oblique + * wall. The coordinate sets stay raw - premultiplying + * them by 1/w and dividing by the set's own third + * float also renders (a measured positive) but broke + * the sky, whose 1/w is small enough that the scaled + * coordinates lose their precision; the plane form is + * the vendor's and keeps every float in range. + * SGX_NO_PERSP restores the affine 1.0. */ + if (src[e].rhw) { + memcpy(&oow, p + 12, 4); + if (oow > 0.0f && oow <= 1e30f) + o[3] = oow; + } + } + /* The hardware path uploads pre-transform attributes, so the + * origin has to come off the viewport translate the vertex + * program applies last - taking it off the attribute shifts a + * coordinate that has not been transformed yet. + * + * Only while the viewport rides in the record: under the MTE + * scheme it is in group 8, ctx_set_mte_viewport() takes the + * origin off it there, and these dwords are a third + * attribute's. */ + if (ctx->hwtcl && ctx->dmg_on > 0 && !sgx_mte_viewport()) { + o[SGX_HWTCL_VP_AO + 1] -= (float)ctx->dmg_x0; + o[SGX_HWTCL_VP_AO + 3] -= (float)ctx->dmg_y0; + } + memcpy(out + (size_t)k * ctx->vtx_floats, o, + ctx->vtx_floats * sizeof *o); + if (!ctx->draw_idx) + idx[k] = (uint16_t)(first_v + k); + } + /* The caller's own indices, moved onto this draw's vertices. Its + * values name records within the draw, so first_v is added to each - + * the same offset the sequential form applies. */ + if (ctx->draw_idx) + for (k = 0; k < nidx; k++) + idx[k] = (uint16_t)(first_v + ctx->draw_idx[k]); + + /* Which record the frame will read as degenerate. A vertex whose + * position is all zero reaches the rasteriser with w of zero and + * draws a sliver to the corner of the viewport; this says whether + * one is in the block and where, so the search is not by picture. */ + if (SGX_ENV("SGX_VTX_SCAN")) { + unsigned q, nz = 0, first = 0, last = 0; + + for (q = 0; q < count; q++) { + const float *o = out + (size_t)q * ctx->vtx_floats; + + if (o[0] != 0.0f || o[1] != 0.0f || + o[2] != 0.0f || o[3] != 0.0f) + continue; + if (!nz) + first = q; + last = q; + nz++; + } + if (nz) + fprintf(sgx_log(), "sgx: vtx scan: %u of %u record(s) " + "have a zero position, first %u last %u " + "(base %u, first_v %u, floats %u)\n", + nz, count, first, last, base, + (unsigned)first_v, ctx->vtx_floats); + else + fprintf(sgx_log(), "sgx: vtx scan: none of %u\n", + count); + } + + /* The record as it was written, not as it arrived. Every dump this + * driver has of a draw-module vertex prints the draw module's own + * buffer - sgx_vbuf_emit_raw()'s raw[] and col() - and the elements + * are applied after it, so a value lost in the conversion looks + * exactly like one that was never there. SGX_DUMP_RECORD prints what + * the frame will actually be read from, which is the other end of + * that comparison. */ + if (SGX_ENV("SGX_DUMP_RECORD")) { + const char *rn = SGX_ENVS("SGX_DUMP_RECORD_N"); + unsigned lim = rn && *rn ? (unsigned)atoi(rn) : 6u; + unsigned q, e; + + for (q = 0; q < count && q < lim; q++) { + const float *o = out + (size_t)q * ctx->vtx_floats; + + fprintf(sgx_log(), "sgx: record vtx %u:", base + q); + for (e = 0; e < ctx->vtx_floats; e++) + fprintf(sgx_log(), " %g", o[e]); + fputc('\n', sgx_log()); + } + } + + if (bx1 >= bx0) { + ctx->last_box[0] = bx0; ctx->last_box[1] = by0; + ctx->last_box[2] = bx1; ctx->last_box[3] = by1; + } + if (trace && bx1 >= bx0) + fprintf(sgx_log(), "sgx: dmg trace: box %.0f,%.0f..%.0f,%.0f " + "dmg_on %d region %ux%u at %u,%u hwtcl %d n %u\n", + bx0, by0, bx1, by1, ctx->dmg_on, ctx->dmg_w, + ctx->dmg_h, ctx->dmg_x0, ctx->dmg_y0, ctx->hwtcl, + count); + ctx->vtx_uploaded = base + nidx; + ctx->nidx = ctx->vtx_uploaded; + /* Extend the open record rather than opening another: draws that share + * their state share a primitive block. */ + if (ctx->range_open && ctx->nrange) + ctx->range[ctx->nrange - 1].count += nidx; + else if (ctx_range_room(ctx)) { + /* The vertices are already written and the frame's records + * cannot describe them, so the frame is wrong whatever + * happens next. Saying so beats submitting it and losing the + * geometry silently. */ + ctx->range_dropped++; + sgx_ctx_dbg("upload: no room for another record\n"); + return ctx_split("no room for another record"); + } + else { + ctx->range[ctx->nrange].first = base; + ctx->range[ctx->nrange].count = nidx; + ctx->range[ctx->nrange].uni = ctx->hwtcl ? ctx_uni_set(ctx) : 0; + ctx->range[ctx->nrange].vpds_prog = + (ctx->hwtcl && ctx->vs && ctx->vs->nlimm) ? 1 : 0; + ctx->range[ctx->nrange].hwtcl = ctx->hwtcl ? 1 : 0; + ctx->range[ctx->nrange].vtx_floats = ctx->vtx_floats; + if (ctx->range[ctx->nrange].uni >= SGX_HWTCL_MAX_UNI_SETS) { + sgx_ctx_dbg("upload: vertex uniform sets spent\n"); + return ctx_split("vertex uniform sets spent"); + } + ctx->range[ctx->nrange].fs_uni = ctx_fs_uni_set(ctx); + if (ctx->range[ctx->nrange].fs_uni >= SGX_FS_MAX_UNI_SETS) { + sgx_ctx_dbg("upload: fragment uniform sets spent " + "(%u)\n", ctx->range[ctx->nrange].fs_uni); + return ctx_split("fragment uniform sets spent"); + } + /* The state as it is now: a record carries its own, so what + * matters is what was bound when the group opened. */ + ctx_update_state(ctx); + ctx_isp_words(ctx, &ctx->range[ctx->nrange].isp, + &ctx->range[ctx->nrange].isp_bf); + ctx_cull_word(ctx, &ctx->range[ctx->nrange].cull, + &ctx->range[ctx->nrange].cull_on); + /* The shape this block rasterises, so a frame that changes + * shape opens another record instead of ending. */ + ctx->range[ctx->nrange].objtype = ctx->prim_objtype; + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: range %u isp %08x b %08x c %08x " + "bf %08x/%08x/%08x (blend %d, " + "kill %d, passtype %u)\n", ctx->nrange, + ctx->range[ctx->nrange].isp.a, + ctx->range[ctx->nrange].isp.b, + ctx->range[ctx->nrange].isp.c, + ctx->range[ctx->nrange].isp_bf.a, + ctx->range[ctx->nrange].isp_bf.b, + ctx->range[ctx->nrange].isp_bf.c, + ctx->blend_translucent, + ctx->fs ? (int)ctx->fs->uses_kill : -1, + ctx_isp_passtype(ctx)); + ctx->range[ctx->nrange].fs = ctx->fs; + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: range %u opens with fs %p, " + "%u set(s)\n", ctx->nrange, + (const void *)ctx->fs, + ctx->fs ? ctx->fs->attribs.nset : 0); + ctx->range[ctx->nrange].blend = ctx->blend_on ? ctx->blend_op : + SGX_BLEND_NONE; + ctx->range[ctx->nrange].blend_desc = ctx->blend_desc; + ctx->range[ctx->nrange].blend_const = ctx->blend_const; + ctx->range[ctx->nrange].blend_insn = + ctx->blend_on ? ctx->blend_insn : 0; + ctx->range[ctx->nrange].blended = ctx->blend_on; + ctx->range[ctx->nrange].logicop = ctx->logicop; + ctx->range[ctx->nrange].logicop_func = ctx->logicop_func; + /* Every unit, described the same way. Cleared first: these + * are only ever set, so a stale one sent an untextured frame + * down the per-record path with a bogus descriptor. + * + * A bound view with no address is not a texture. The words + * describe the format and the size and say nothing about the + * base, so a view whose buffer never bound was issued as a + * descriptor pointing at zero - and the render faulted there + * with the texture cache as the requestor, which stopped the + * core and refused every submit after it. */ + for (u = 0; u < SGX_MAX_TEX_UNITS; u++) { + struct sgx_bo *tbo = ctx->tex_bo[u]; + uint32_t w0 = 0, w1 = 0; + enum sgx_sampler_status ss; + + memset(&ctx->range[ctx->nrange].tex[u], 0, + sizeof ctx->range[ctx->nrange].tex[u]); + if (!(ctx->view_bound & (1u << u)) || + !ctx->views[u].gpu_va) { + if (u && SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: unit %u not " + "bound: view_bound 0x%x, va " + "0x%x\n", u, ctx->view_bound, + (unsigned)ctx->views[u].gpu_va); + continue; + } + ss = sgx_sampler_words(&ctx->views[u], + &ctx->samplers[u], &w0, &w1); + /* A unit the program samples but the record cannot + * describe reads an all-zero descriptor, which comes + * back black - so why it could not is said rather than + * left to the picture. */ + if (ss != SGX_SAMPLER_OK) { + fprintf(sgx_log(), "sgx: unit %u: no descriptor " + "(%d); the record samples it black\n", + u, (int)ss); + continue; + } + ctx->range[ctx->nrange].tex[u].w0 = w0; + ctx->range[ctx->nrange].tex[u].w1 = w1; + ctx->range[ctx->nrange].tex[u].va = + sgx_sampler_view_base(&ctx->views[u], + &ctx->samplers[u]); + ctx->range[ctx->nrange].tex[u].nchunks = + ctx->views[u].nchunks; + ctx->range[ctx->nrange].tex[u].chunk_size = + ctx->views[u].chunk_size; + ctx->range[ctx->nrange].tex[u].obj = tbo; + ctx->range[ctx->nrange].tex[u].bound = 1; + /* Its own cue to end the frame, which was dropped + * here: past the submit's texture budget the record + * still named an address whose object never entered + * the list, and the core faulted on it. The range is + * not committed yet, so the caller's flush and retry + * rebuilds it. */ + if (sgx_frame_add_tex(ctx, tbo) == -ENOSPC) { + sgx_ctx_dbg("upload: texture budget spent on " + "unit %u\n", u); + return ctx_split("texture budget spent"); + } + } + ctx->nrange++; + ctx->range_open = 1; + } + *(uint32_t *)(b + ctx->draw_cmd_off) = + sgx_ctx_draw_cmd(ctx, ctx->nidx); + return 0; +} + +void sgx_next_draw_record(struct sgx_context *ctx) +{ + if (ctx) + ctx->range_open = 0; +} + +int sgx_set_vistest(struct sgx_context *ctx, int reg) +{ + if (!ctx) + return -EINVAL; + if (reg >= 0 && (unsigned)reg >= SGX_ISP_VISTEST_REGS) + return -EINVAL; + if (ctx->vis_reg == reg) + return 0; + ctx->vis_reg = reg; + /* The draws already in the open record were made under the old + * setting, and a record carries one ISP word for all of them. */ + sgx_next_draw_record(ctx); + return 0; +} + +int sgx_frame_has_vistest(const struct sgx_context *ctx) +{ + unsigned k; + + if (!ctx) + return 0; + for (k = 0; k < ctx->nrange; k++) + if (ctx->range[k].isp.a & SGX_ISP_BPRES && + ctx->range[k].isp.b & SGX_ISPB_VISTEST) + return 1; + return 0; +} + +/* Whether the scissor actually clips anything. A box that covers the surface + * is not a scissor: it removes no fragment, but taking it as one puts every + * draw on the draw module - the part's transform is given up because only the + * draw module clips - and clips every triangle against it on the CPU. A + * fullscreen client that scissors to its own viewport paid both. */ +int sgx_scissor_clips(const struct sgx_context *ctx) +{ + if (!ctx || !ctx->scissor_on || !ctx->scissor_enable) + return 0; + return ctx->scissor_x0 > 0.0f || ctx->scissor_y0 > 0.0f || + ctx->scissor_x1 < (float)ctx->fb.width || + ctx->scissor_y1 < (float)ctx->fb.height; +} + +/* Whether the frame will really be built for its rectangle. The geometry is + * moved into the rectangle as each draw is uploaded, so this has to answer the + * same at the draw as at the flush: a frame that clears, or that loads depth + * back, renders the whole surface, and moved geometry in it lands at the + * origin instead of in the rectangle. */ +static int ctx_damage_usable(const struct sgx_context *ctx) +{ + if (ctx->clear_on) + return 0; + return !ctx->depth_stored || ctx->zclear_on || SGX_ENV("SGX_NO_ZLOAD"); +} + +/* Triangle corners the frame can still take, rounded down to a whole + * triangle. Zero means the frame has to go out first. + * + * A draw larger than a frame holds cannot be made smaller by splitting - the + * retry offers the same call again and it is refused again, which is how + * glmark2's refract scene, 69666 triangles in one call against a heap sized + * for 25600, drew nothing at all. The caller uses this to hand the draw over + * in pieces instead. The four bounds are the ones sgx_draw() and + * ctx_upload_vertices() apply, so a piece this reports fits all of them. */ +unsigned sgx_draw_room(const struct sgx_context *ctx) +{ + size_t room = (size_t)-1, s, cur, cap; + + if (!ctx) + return 0; + if (ctx->max_prims) { + size_t used = ctx->nidx; + + room = (size_t)ctx->max_prims * 3u; + room = room > used ? room - used : 0; + } + /* Where the vertices go, in floats, against where the indices start */ + s = ctx->vtx_floats ? ctx->vtx_floats : 1u; + cur = (ctx->vtx_float_cursor + s - 1) / s * s; + cap = ((size_t)XPSB_IDX_OFF - XPSB_VTX_OFF) / 4u; + if (cur + s > cap) + return 0; + if ((cap - cur) / s < room) + room = (cap - cur) / s; + /* The index array behind them */ + cap = ctx->bo[SGX_CTX_BO_VTX].size > (size_t)XPSB_IDX_OFF ? + (ctx->bo[SGX_CTX_BO_VTX].size - XPSB_IDX_OFF) / 2u : 0; + if (cap <= ctx->vtx_uploaded) + return 0; + if (cap - ctx->vtx_uploaded < room) + room = cap - ctx->vtx_uploaded; + /* And what a sixteen-bit index reaches */ + cur = cur / s; + if (cur >= 0xffffu) + return 0; + if (0xffffu - cur < room) + room = 0xffffu - cur; + return (unsigned)(room - room % 3u); +} + +int sgx_draw(struct sgx_context *ctx, unsigned vertex_count) +{ + if (!ctx || !vertex_count) + return -EINVAL; + if (!ctx->cookie_valid) + return -EINVAL; /* no framebuffer bound */ + /* The parameter heap is sized for max_prims triangles and the DPM is + * told that figure once per boot, so a frame that bins more than it + * runs the tiler off the end of the heap - which faults and stops the + * core rather than failing the frame. Refusing the draw is the only + * thing that keeps a client that draws too much from taking the + * machine with it. */ + if (ctx->max_prims && + (ctx->nidx + vertex_count) / 3u > ctx->max_prims) { + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), + "sgx: draw refused: %u triangles is past the " + "%u the parameter heap is sized for\n", + (ctx->nidx + vertex_count) / 3u, ctx->max_prims); + return ctx_split("past the triangles the heap is sized for"); + } + /* SGX_SPLIT_EVERY ends the frame at every draw that has one under it, + * with no damage involved. It is how the defect below was shown not to + * belong to the damage regions at all: with the regions off and this + * on, x_prims loses its fill in five clients of six, the same as with + * the regions on and this off. */ + if (ctx->draws && SGX_ENV("SGX_SPLIT_EVERY")) + return ctx_split("forced every draw"); + /* The rectangle this frame is built for, taken from the first draw + * that has one. A later draw wanting a different one gives it up and + * the frame goes back to the whole surface, because the geometry + * already written was moved into the first. */ + /* On, now that it draws in the right place. It was off because a + * rectangle whose origin was not 0,0 drew its content at pos minus + * that origin - an xterm at +200+300 landed near the top-left corner + * - and that was the same defect as the ignored clip: the geometry is + * moved into the rectangle by sgx_vbuf_emit(), which is the draw + * module's path, and the part's own transform skipped the move + * whenever the MTE held the viewport. A damaged frame needs a scissor + * and a scissor now keeps the transform off, so the move always + * happens. + * + * Checked against a server with no acceleration: eighteen of the + * nineteen requests in x_prims match exactly with this on, and the + * nineteenth - a bitmap clip - is closer with it than without. Window + * Maker's desktop and its window buttons are right, and the corner + * that held the black bar has 305 black pixels against the + * unaccelerated server's 307. + * + * Worth having: the twm menu goes from 831 ms to 525, ioquake3 from + * 9.6 fps to 12.8, and the feature suite, glmark2-drm build, texture + * and shading do not move. SGX_NO_DAMAGE puts it back. */ + if (ctx->scissor_on && ctx->scissor_enable && ctx_damage_usable(ctx) && + !SGX_ENV("SGX_NO_DAMAGE")) { + unsigned x0 = (unsigned)(ctx->scissor_x0 < 0.0f ? 0.0f : + ctx->scissor_x0) / 16u * 16u; + unsigned y0 = (unsigned)(ctx->scissor_y0 < 0.0f ? 0.0f : + ctx->scissor_y0) / 16u * 16u; + float fx1 = ctx->scissor_x1, fy1 = ctx->scissor_y1; + unsigned x1, y1; + + if (fx1 > (float)ctx->fb.width) fx1 = (float)ctx->fb.width; + if (fy1 > (float)ctx->fb.height) fy1 = (float)ctx->fb.height; + x1 = (((unsigned)fx1 + 15u) / 16u) * 16u; + y1 = (((unsigned)fy1 + 15u) / 16u) * 16u; + if (x1 > ctx->fb.width) x1 = ctx->fb.width; + if (y1 > ctx->fb.height) y1 = ctx->fb.height; + /* A rectangle that covers the surface is not a damage + * rectangle: it saves the pass nothing and it makes every + * unscissored draw after it end the frame, because the + * geometry already in went in moved. A fullscreen client that + * scissors to its own viewport hit that once a frame - + * ioquake3 submitted two frames per swap for it. */ + if (x1 > x0 && y1 > y0 && + !(x0 == 0 && y0 == 0 && + x1 >= ctx->fb.width && y1 >= ctx->fb.height)) { + if (!ctx->dmg_on) { + ctx->dmg_on = 1; + ctx->dmg_x0 = x0; + ctx->dmg_y0 = y0; + ctx->dmg_w = x1 - x0; + ctx->dmg_h = y1 - y0; + } else if (ctx->dmg_on > 0 && + (ctx->dmg_x0 != x0 || ctx->dmg_y0 != y0 || + ctx->dmg_w != x1 - x0 || + ctx->dmg_h != y1 - y0)) { + /* Only a frame that is being built for a + * rectangle can have that rectangle move. One + * that gave rectangles up renders the whole + * surface and its geometry went in unmoved, + * so a scissor that differs from the stale + * corner is nothing to it - ending the frame + * there cost ioquake3 a second submit of + * every frame it drew. */ + /* One frame renders one region, and every + * vertex already in the buffer was written + * relative to its corner. Giving the region + * up here left those draws shifted by the old + * corner while everything after was written + * without one - half a frame displaced toward + * the origin, which rasterises as wedges + * running back to the top left. A scrolling + * list repaints through a succession of + * regions, so it met this every time. + * + * The frame ends instead; the continuation + * loads each tile back, so what is already + * drawn survives. */ + if (ctx->draws && + !SGX_ENV("SGX_NO_DAMAGE_SPLIT")) { + sgx_ctx_dbg("damage: the region moved " + "to %u,%u; the frame " + "ends\n", x0, y0); + return ctx_split("the damage region moved"); + } + ctx->dmg_x0 = x0; + ctx->dmg_y0 = y0; + ctx->dmg_w = x1 - x0; + ctx->dmg_h = y1 - y0; + } + } else { + if (ctx->dmg_on > 0 && ctx->draws && + !SGX_ENV("SGX_NO_DAMAGE_SPLIT")) { + if (ctx->dmg_x0 || ctx->dmg_y0) { + sgx_ctx_dbg("damage: no region now; " + "the frame ends\n"); + return ctx_split("no damage region now"); + } + ctx->dmg_w = ctx->fb.width; + ctx->dmg_h = ctx->fb.height; + } else + ctx->dmg_on = -1; + } + } else { + if (ctx->dmg_on > 0 && ctx->draws && + !SGX_ENV("SGX_NO_DAMAGE_SPLIT")) { + /* The draw reaches the whole surface. From the origin + * the extent can grow to it and the frame carries on; + * anywhere else the corner would have to move. */ + if (ctx->dmg_x0 || ctx->dmg_y0) { + sgx_ctx_dbg("damage: unscissored draw; the " + "frame ends\n"); + return ctx_split("an unscissored draw"); + } + ctx->dmg_w = ctx->fb.width; + ctx->dmg_h = ctx->fb.height; + } else + /* Also when it is the frame's first draw. Its vertices + * go in at absolute coordinates, so a later scissored + * draw must not move the frame into a rectangle they + * were never written for - that dropped every glyph an + * xterm painted. */ + ctx->dmg_on = -1; + } + + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: draw: %u vert(s), fs %p, %u set(s)\n", + vertex_count, (const void *)ctx->fs, + ctx->fs ? ctx->fs->attribs.nset : 0); + /* No fixed-function path exists on this hardware, so a draw without + * both programs has nothing to run. */ + if (!ctx->vs || !ctx->fs) + return -EINVAL; + /* A buffer that lost its address reads as zero on the part, and the + * vertex fetch faults there and stops the core rather than failing the + * frame - which under X takes every later frame with it. */ + if (!ctx->bo[SGX_CTX_BO_TARGET].gpu_va) { + /* Said every time, not only under SGX_DEBUG. A frame dropped + * for this reason is a window that stays blank, and the whole + * of what distinguishes it from every other blank window is + * this line. */ + static int said; + + if (!said) { + said = 1; + fprintf(sgx_log(), "sgx: draw refused: the render " + "target has no address - what it would have " + "drawn will not appear\n"); + } + return -EFAULT; + } + /* The frame samples one unit whatever the program does, so the unit + * has to hold an address even when nothing is bound to it. */ + if (!ctx->bo[SGX_CTX_BO_TEX].gpu_va) { + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: draw refused: texture unit 0 has " + "no address\n"); + return -EFAULT; + } + ctx_update_state(ctx); + /* One frame, one record width. The vertex DMA's dword count, the byte + * stride the fetch walks and the group-10 field all sit outside the + * state block a record copies, so they cannot be made this record's + * own - and vertices already written at the old width would be read + * at the new one. A draw that needs a different width ends the frame + * instead; -ENOSPC is the caller's cue, and its continuation loads + * each tile back so the draws already in the frame survive. */ + /* A draw whose record is a different width joins the frame at the + * width the frame already has, rather than ending it. ctx_emit_samplers() + * already forces the frame's width onto whatever program is bound at + * flush, so ending the frame here was a second guard on a width that + * had been settled anyway - and it cost the draws already in the + * frame: gtk3-demo's list pane came out empty but for the last two + * rows, and renders in full without the split. The feature suite is + * identical either way, 60 of 67, so the split was buying nothing + * measurable. SGX_WIDTH_SPLIT restores it. */ + /* Only onto a wider frame. The floats past what this program reads are + * simply not read, but a frame narrower than the record the program + * asks for truncates its last attributes - measured as an iterated set + * arriving with its fourth component zero, and as a second set reading + * the first one's registers. That ends the frame below instead. */ + if (ctx->draws && ctx->frame_vtx_floats && + ctx->frame_vtx_floats > ctx->vtx_floats && + !SGX_ENV("SGX_WIDTH_SPLIT")) + ctx->vtx_floats = ctx->frame_vtx_floats; + /* A wider record cannot join: the vertices already in the buffer were + * written at the frame's width and cannot be re-read at another, and + * the fetch's DMA control and byte stride are frame-global - a record + * carries its own only in its vertex PDS copy, and the frame's own + * fields end up holding whatever the last record set. So an eleven + * float record followed by a thirteen leaves the eleven fetched at + * thirteen: its w arrives as another vertex's data and the triangles + * land far from where they were drawn. That is the white rectangle + * beside an xterm's menu. + * + * Only wider. A narrower record is widened onto the frame above, + * which writes its vertices at the frame's width and is safe. + * SGX_NO_WIDTH_SPLIT keeps them in one frame for anyone working on + * making the per-record width actually carry. */ + if (ctx->draws && ctx->frame_vtx_floats && + ctx->frame_vtx_floats != ctx->vtx_floats && + !SGX_ENV("SGX_NO_WIDTH_SPLIT")) { + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: draw: the record is %u floats " + "and the frame is %u, so the frame ends\n", + ctx->vtx_floats, ctx->frame_vtx_floats); + sgx_perf_note("record width"); + return ctx_split("the record is another width"); + } + { + uint64_t tv = sgx_perf_mark(); + int ret = ctx_upload_vertices(ctx, vertex_count); + + sgx_perf_add(3, tv); + + if (ret) + return ret; + ctx->frame_vtx_floats = ctx->vtx_floats; + } + /* One draw's worth, whatever happened to it: a retry after a split + * re-enters through sgx_draw() and must not pick these up a second + * time, and neither must the draw after. */ + ctx->draw_idx = NULL; + ctx->draw_nidx = 0; + ctx->draws++; + return 0; +} + +unsigned sgx_context_ranges(const struct sgx_context *ctx) +{ + return ctx ? ctx->nrange : 0; +} + +/* Everything a frame accumulates, put back. Used by both ends of a flush: + * a frame that submitted is finished with, and a frame that was refused must + * not be carried into the next one. */ +/* Forget a shader that is about to be freed. The context keeps the bound + * program, and each draw record opened this frame kept the one that was bound + * when it opened - all as raw pointers - so a shader deleted mid-frame has to + * be cleared out of every one of them or the upload reads freed memory. */ +void sgx_forget_fs(struct sgx_context *ctx, const void *fs) +{ + unsigned k; + + if (!ctx || !fs) + return; + if (ctx->fs == fs) + ctx->fs = NULL; + for (k = 0; k < ctx->nrange; k++) + if (ctx->range[k].fs == fs) { + ctx->range[k].fs = NULL; + /* The frame can no longer be built as described. */ + sgx_ctx_dbg("forget fs: record %u still names the " + "program being deleted; the frame is " + "dropped\n", k); + ctx->frame_invalid = 1; + } +} + +void sgx_forget_vs(struct sgx_context *ctx, const void *vs) +{ + if (!ctx || !vs) + return; + if (ctx->vs == vs) { + ctx->vs = NULL; + ctx->frame_invalid = 1; + } +} + +void sgx_context_drop_frame(struct sgx_context *ctx) +{ + if (!ctx) + return; + ctx->draws = 0; + ctx->clear_pending = 0; + ctx->nrange = 0; + ctx->nconst_copy = 0; + ctx->frame_vtx_floats = 0; + ctx->nframe_tex = 0; + ctx->nuni_set = 0; + ctx->nfs_uni_set = 0; + ctx->range_dropped = 0; + ctx->range_open = 0; + ctx->vtx_uploaded = 0; /* the next frame starts its own vertices */ + ctx->vtx_float_cursor = 0; + ctx->nidx = 0; + ctx->frame_invalid = 0; + /* The clear belonged to the frame that just went out. GL's clear is a + * command, not state, so the frame after it loads the target back + * instead of clearing again - which is what every caller that flushes + * mid-frame already says by hand. + * + * Left set, a pass that draws into a target another pass had cleared + * cleared it again: glmark2's terrain blends a bloom over the terrain + * it just rendered, and wiped it instead. The overlay's own output was + * all that reached the screen. */ + ctx->clear_on = 0; + ctx->zclear_on = 0; +} + +unsigned sgx_context_dropped(const struct sgx_context *ctx) +{ + return ctx ? ctx->range_dropped : 0; +} + +/* What the frame is actually going to draw: one line per primitive block, so + * a record that covers one triangle where it should cover a strip is visible + * rather than inferred. */ +void sgx_context_dump_ranges(const struct sgx_context *ctx) +{ + unsigned k, tot = 0; + + if (!ctx || !SGX_ENV("SGX_DEBUG")) + return; + for (k = 0; k < ctx->nrange; k++) + tot += ctx->range[k].count; + fprintf(sgx_log(), "sgx: %u record(s), %u vertices total\n", + ctx->nrange, tot); + for (k = 0; k < ctx->nrange && k < 8; k++) + fprintf(sgx_log(), "sgx: record %u: first %u count %u isp %08x\n", + k, ctx->range[k].first, ctx->range[k].count, + ctx->range[k].isp.a); + /* And where the geometry actually is. Every other field of a record + * can be right while its vertices are somewhere the tile never sees, + * and nothing printed them - so a draw that produces no pixels looked + * identical to one that produces them. The first vertex of each + * record, as the floats the hardware will read. */ + if (SGX_ENV("SGX_DUMP_VTX")) { + const char *vb = ctx->bo[SGX_CTX_BO_VTX].map; + const float *v = vb ? (const float *)(vb + XPSB_VTX_OFF) : + NULL; + unsigned stride = ctx->vtx_floats; + + for (k = 0; v && stride && k < ctx->nrange && k < 8; k++) { + const float *p = v + (size_t)ctx->range[k].first * + stride; + unsigned q; + + const uint16_t *ix = (const uint16_t *) + (vb + XPSB_IDX_OFF); + + fprintf(sgx_log(), "sgx: vtx %u (stride %u):", k, + stride); + for (q = 0; q < stride && q < 8; q++) + fprintf(sgx_log(), " %.3f", p[q]); + /* And the indices the record actually draws through: + * correct vertices reached by a wrong index list + * rasterise nothing, and look identical to every + * other field being right. */ + fprintf(sgx_log(), " idx:"); + for (q = 0; q < 6 && q < ctx->range[k].count; q++) + fprintf(sgx_log(), " %u", + ix[ctx->range[k].first + q]); + fputc('\n', stderr); + } + } +} + +/* Where the fragment program's constants sit in the heap before the secondary + * PDS program moves them into the secondary attribute bank. The heap has + * nothing between 0x400 and 0x43e4 that any relocation names; 0x400 is the + * secondary program itself and 0x480 and 0x500 are the hardware-transform + * path's, so this is the next free block. */ +/* movs doutd, ds0[0], ds0[1], .. - the same instruction the vertex constant + * loader uses, reading the two data dwords written below it. */ +#define SGX_PDS_DOUTD_0 0x07030223u + +/* Copy each bound program's literal pool into the heap object. The secondary + * PDS program DMAs it from there into the sa bank; see work/uniform-abi/. */ +/* Put the compiled fragment program where the frame's binding words already + * point. The captured frame carries its own program in that slot; sgxtri + * replaces it the same way for its Gouraud path, so a compiled one that fits + * the slot runs in its place. */ +static int ctx_blend_program(const struct sgx_shader *fs, enum sgx_blend_op op, + const struct sgx_blend_desc *d, uint32_t constant, + int have_insn, uint64_t *out, unsigned cap, + unsigned *n, unsigned *ntemps, unsigned logicop, + unsigned logicop_func); + +/* Write a fragment program into the USSE buffer, in the inline slot when it + * fits and in the long region when it does not. + * + * The captured frame's fragment slot ends where the vertex program begins, + * which leaves room for eight instructions. A blended program was written + * there unconditionally and without a length check, so a program of any real + * size ran off the end of the slot and over the vertex program behind it - and + * a blended draw whose fragment program also sampled did exactly that. */ +static int ctx_write_fs_code(struct sgx_context *ctx, const uint64_t *code, + unsigned n) +{ + unsigned small = (XPSB_USSE_VTX_OFF - XPSB_USSE_FRAG_OFF) / 4; + uint32_t uo = XPSB_USSE_FRAG_OFF; + unsigned room = small, i; + uint32_t *slot; + + if (n * 2u > small || SGX_ENV("SGX_FS_FORCE_LONG")) { + uo = SGX_USSE_FRAG_LONG; + room = (SGX_USSE_FRAG_LONG_END - SGX_USSE_FRAG_LONG) / 4; + if (n * 2u > room) + return -ENOSPC; + } + if (uo + room * 4u > ctx->bo[SGX_CTX_BO_USSE].size) + return -ENOSPC; + slot = (uint32_t *)((char *)ctx->bo[SGX_CTX_BO_USSE].map + uo); + memset(slot, 0, room * 4); + for (i = 0; i < n; i++) { + slot[i * 2] = (uint32_t)(code[i] & 0xffffffffu); + slot[i * 2 + 1] = (uint32_t)(code[i] >> 32); + } + /* Where it ended up, for the relocation to name it by. + * + * ctx_apply_relocs() runs after this and rebuilds the dword for offset + * 0x100 unconditionally, so patching the word here achieved nothing + * and the frame went on executing the captured program - which is why + * a shader too long for the inline slot came back as a texel of the + * default texture. The frame generator relocates the fragment program + * from frag_use_off; it was simply never given a value. */ + ctx->fs_use_off = (uo != XPSB_USSE_FRAG_OFF) ? uo : 0u; + ctx->fs_use_size = ctx->fs_use_off ? n * 8u : 0u; + return 0; +} + +static int ctx_upload_fs(struct sgx_context *ctx) +{ + uint32_t *slot; + unsigned i; + /* The first group's, not the last state bound: the groups after it + * carry their own program, and this one is theirs. Held locally - + * assigning it to ctx->fs left the caller's bound program replaced by + * record 0's for every frame that followed. */ + const struct sgx_shader *fs = sgx_frame_fs(ctx); + + if (!ctx->bo[SGX_CTX_BO_USSE].map) + return -EINVAL; + + { + enum sgx_blend_op op = ctx->nrange ? ctx->range[0].blend : + (ctx->blend_on ? ctx->blend_op : + SGX_BLEND_NONE); + int blended = ctx->nrange ? ctx->range[0].blended : + (op != SGX_BLEND_NONE && op != SGX_BLEND_SRC); + + /* The blend is appended to the bound program, so with no + * program there is nothing to append to and nothing to + * upload - blended or not. Testing only the unblended case + * dereferenced a null shader whenever a program was refused + * while blending was on. */ + if (!fs || !fs->compiled) + return 0; + if (blended) { + /* Heap, not stack: the code buffer grows with the + * program now, and SGX_SHADER_MAX_CODE entries of + * eight bytes is far past what a frame is worth + * putting on the stack. */ + /* Past the program: a blend whose source came from + * an attribute keeps the move and adds its list + * behind it. */ + unsigned cap = fs->ninsns + 1u + SGX_BLEND_MAX_INSNS; + uint64_t *code = malloc((size_t)cap * sizeof *code); + unsigned n = 0, nt = 0, i; + int r; + + if (!code) + return -ENOMEM; + r = ctx_blend_program(fs, op, &ctx->blend_desc, + ctx->blend_const, ctx->blend_insn, + code, cap, &n, &nt, ctx->logicop, + ctx->logicop_func); + /* An operator this cannot build draws unblended - + * visibly wrong, but the geometry is there. Failing + * here fails the flush instead, and the whole frame + * goes with it: ioquake3's intro lost 261 frames that + * way and showed nothing at all. The per-record path + * further down already degrades exactly like this. */ + if (r) { + sgx_ctx_dbg("blend: this program cannot carry " + "the operator (%d), so the draw is " + "unblended rather than lost\n", r); + free(code); + } else { + r = ctx_write_fs_code(ctx, code, n); + free(code); + if (r) + return r; + ctx->fs_uploaded = n; + ctx->fs_ntemps = nt; + goto temps; + } + } + } + + if (0) { + struct xpsb_shader_flags fl; + uint32_t w[4]; + + memset(&fl, 0, sizeof fl); + fl.scalar_mask = 1; /* no mask picture: the source alone */ + if (xpsb_composite_shader((int)ctx->blend_op, &fl, w)) + return -EINVAL; + slot = (uint32_t *)((char *)ctx->bo[SGX_CTX_BO_USSE].map + + XPSB_USSE_FRAG_OFF); + memset(slot, 0, XPSB_USSE_SLOT_DW * 4); + memcpy(slot, w, sizeof w); + ctx->fs_uploaded = 2; + return 0; + } + + if (!fs || !fs->compiled) + return 0; + /* The captured frame's fragment slot ends where the vertex program + * begins, which leaves room for eight instructions. That is enough for + * the programs this driver started with and nowhere near enough for a + * real one, so a longer program goes into the space above the captured + * image instead and the PDS word that names it is rewritten to point + * there. The short case keeps the captured layout untouched. */ + { + int r = ctx_write_fs_code(ctx, fs->code, fs->ninsns); + + if (r) + return r; + if (SGX_ENV("SGX_DEBUG")) { + uint32_t *heap = ctx->bo[SGX_CTX_BO_HEAP].map; + uint32_t w0 = 0, w1 = 0; + + ctx_use_word(ctx, XPSB_USSE_FRAG_OFF, &w0); + ctx_use_word(ctx, SGX_USSE_FRAG_LONG, &w1); + fprintf(sgx_log(), "sgx: use word: frame has 0x%08x, " + "computed for 0x%x is 0x%08x, for 0x%x is " + "0x%08x\n", heap ? heap[XPSB_PRI_PDS_DW] : 0, + XPSB_USSE_FRAG_OFF, w0, SGX_USSE_FRAG_LONG, w1); + } + } + ctx->fs_uploaded = fs->ninsns; + /* Nothing was appended, so what the codegen counted is what runs. */ + ctx->fs_ntemps = fs->ntemps; +temps: + { + uint32_t *heap = ctx->bo[SGX_CTX_BO_HEAP].map; + /* The task is given what the code in the slot references, + * counted while it was generated. Adding one for a blend + * whether or not the blend appended anything, and adding it + * again to a program that was never blended, was a guess in + * both directions. */ + unsigned t = fs ? ctx->fs_ntemps : 0; + unsigned n, hi; + const char *e = SGX_ENVS("SGX_TEMPS"); + + if (!sgx_temp_ok(t)) + return -ENOSPC; + if (e && *e) + t = (unsigned)atoi(e); + n = sgx_temp_field(t); + hi = sgx_temp_field_hi(t); + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: fs temps %u -> field %u, %u " + "insn(s), preiter %u, nuniform %u, pool_base " + "%u, smp_base %u\n", t, n, + fs ? fs->ninsns : 0, + fs ? fs->tex_preiterated : 0, + fs ? fs->nuniform : 0, + fs ? fs->pool_base : 0, + fs ? fs->smp_base : 0); + if (heap) { + heap[XPSB_HEAP_USE_TEMPS] = + (heap[XPSB_HEAP_USE_TEMPS] & ~0xf8000000u) | + ((uint32_t)n << 27); + heap[XPSB_HEAP_USE_TEMPS_HI] = + (heap[XPSB_HEAP_USE_TEMPS_HI] & ~0x3fu) | hi; + } + } + return 0; +} + +/* Build the secondary PDS program that loads the fragment program's constants. + * The frame carries a bare HALT there, so without this a program that reads a + * secondary attribute reads whatever the bank last held - which is why a + * shader with constants rendered wrongly and then stalled the hardware, while + * the same shader without them ran at any length. Returns the data size in + * dwords, or zero when there is nothing to load. */ +static unsigned ctx_sec_pds(struct sgx_context *ctx) +{ + uint32_t *heap = ctx->bo[SGX_CTX_BO_HEAP].map; + const struct sgx_shader *fs = sgx_frame_fs(ctx); + uint32_t ctl, *dms; + unsigned n; + + /* Not only uniforms and constants: a program that samples for itself + * reads the unit's state out of this bank too, and pool_base already + * counts the quad the backend reserves for each sampler. */ + ctx->sec_sa_dwords = 0; + if (!heap || !fs || (!fs->pool_dwords && !fs->nuniform && !fs->pool_base)) + return 0; + /* One transfer for the whole block, from sa0: the uniforms the caller + * set, whatever the frame writes over the sampler slots, and the + * pool. Rounded to a quad so the descriptor's block count has a + * divisor - a prime dword count has none. */ + { + unsigned nuni = fs->nuniform * 4u; + uint32_t *heap2 = heap + SGX_FS_CONST_OFF / 4; + const uint32_t *src = ctx->fs_const; + unsigned ndw = ctx->fs_const_dwords; + + /* Record 0's uniforms, not whatever the caller has bound now. + * The frame serves record 0 and the records behind it get + * copies of their own, so loading this block from the current + * binding gave the first record the last draw's uniforms - + * two quads of different colours both came out the second + * colour, and a client that draws thousands of records left + * every one of them reading the final state. */ + if (ctx->nrange && ctx->fs_uni_set && + ctx->range[0].fs_uni < ctx->nfs_uni_set) { + unsigned set = ctx->range[0].fs_uni; + + if (ctx->fs_uni_set_dwords[set]) { + src = ctx->fs_uni_set[set]; + ndw = ctx->fs_uni_set_dwords[set]; + } + } + if (nuni > ndw) + nuni = ndw; + /* Gathered, not copied: the uniforms are packed to the widths + * the program reads them through, so a flat quad-per-uniform + * write lands on whatever follows - with one scalar uniform + * the literal pool starts at sa1 and the copy took three + * registers of it, which drew the discard test's magenta as + * black. */ + if (nuni) + ctx_pack_uniforms(fs, src, ndw, heap2, + fs->smp_base ? fs->smp_base : + fs->pool_base); + n = fs->pool_base + fs->pool_dwords; + /* And every other record's, not only the frame's. The + * transfer size is one per frame while each record writes its + * own literal pool at its own base, so a record whose pool + * reaches further than record 0's had the tail of it never + * loaded - it read whatever the bank last held. + * ctx_check_record_fits() counted that and nothing acted on + * it. Sizing to the longest costs a larger transfer for the + * frame and makes every record's constants arrive. */ + if (SGX_ENV("SGX_SEC_TRACE")) + fprintf(sgx_log(), "sgx: sec pds: frame program wants " + "%u dword(s), %u range(s) to consider\n", n, + ctx->nrange); + if (ctx->range && !SGX_ENV("SGX_NO_SEC_MAX")) { + unsigned q; + + for (q = 0; q < ctx->nrange; q++) { + const struct sgx_shader *rfs = ctx->range[q].fs; + unsigned want; + + if (!rfs) + continue; + want = rfs->pool_base + rfs->pool_dwords; + if (want > n) { + n = want; + if (SGX_ENV("SGX_SEC_TRACE")) + fprintf(sgx_log(), "sgx: sec " + "pds: record %u wants " + "%u\n", q, want); + } + } + } + /* Rounded to a quad so the descriptor's block count has a + * divisor - a prime dword count has none. */ + n = (n + 3u) & ~3u; + /* A quad is not always enough: the DMA carries at most fifteen + * lines of at most sixteen dwords, and 68, 76, 92 and many + * other multiples of four factor into no such pair - the + * control word could not be built, this returned 0, and the + * frame kept the captured bare HALT, so the program's uniforms + * and literal pool were never loaded and it read whatever the + * bank last held, silently. Every multiple of sixteen up to + * the 240 the encoding reaches does factor, so grow to one + * only when the exact size cannot be expressed - rounding + * every transfer up made the four-dword one the commonest + * programs use into a sixteen-dword one. */ + if (!sgx_hwtcl_dma_ctl(n, 0)) + n = (n + 15u) & ~15u; + if (n > SGX_SEC_PDS_MAX_DW || n > SGX_MAX_FS_CONST_DW) { + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: sec pds: %u dwords is " + "past what one transfer describes\n", n); + return 0; + } + } + ctl = sgx_hwtcl_dma_ctl(n, 0); + if (!ctl) + return 0; + ctx->sec_sa_dwords = n; + /* A four-dword data segment, then the code. The binding's size field + * is EURASIA_PDS_DATASIZE, bits 31:26 in sixteen-byte units, so a data + * segment is a whole unit: two dwords encoded as 2 << 24 declared a + * size of zero with bit 25 - EURASIA_PDS_DEBUG - set, and the code was + * taken to start on the DMA descriptor itself. The vendor's + * PDSGeneratePixelShaderSAProgram() rounds this segment to sixteen + * bytes. Dword 0 is the source address and is a placeholder the + * relocation fills. */ + heap[XPSB_SEC_PDS_OFF / 4 + 0] = 0; + heap[XPSB_SEC_PDS_OFF / 4 + 1] = ctl; + heap[XPSB_SEC_PDS_OFF / 4 + 2] = 0; + heap[XPSB_SEC_PDS_OFF / 4 + 3] = 0; + heap[XPSB_SEC_PDS_OFF / 4 + 4] = SGX_PDS_DOUTD_0; + heap[XPSB_SEC_PDS_OFF / 4 + 5] = XPSB_PDS_HALT; + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: sec pds: %u dwords from sa0 (%u uniform " + "quads, pool at sa%u, %u dwords), ctl 0x%08x, " + "uni %g %g %g %g\n", n, fs->nuniform, fs->pool_base, + fs->pool_dwords, ctl, + ctx->fs_const_dwords > 0 ? *(float *)&ctx->fs_const[0] : 0.0, + ctx->fs_const_dwords > 1 ? *(float *)&ctx->fs_const[1] : 0.0, + ctx->fs_const_dwords > 2 ? *(float *)&ctx->fs_const[2] : 0.0, + ctx->fs_const_dwords > 3 ? *(float *)&ctx->fs_const[3] : 0.0); + /* The bank allocation is in blocks of 128 bytes - thirty-two registers + * - and has to cover what the program reads, not what the captured one + * did. n is a dword count, so it becomes bytes first: rounding the + * dword count itself gave every program under 128 dwords exactly one + * block, which is the thirty-two registers past which the bank was + * measured to corrupt run to run. */ + dms = heap + xpsb_heap_pds_dw(heap, XPSB_PDS_W_CTL); + *dms = (*dms & ~XPSB_SA_BLOCK_MASK) | + ((((n * 4u + 0x7fu) >> 7) << 18) & XPSB_SA_BLOCK_MASK); + /* The two task-size fields beside it. They used to be a division of a + * guessed register pool by the program's own register count, clamped + * to the captured 24 - two numbers with no hardware meaning, and the + * clamp meant the captured value survived for every shader that did + * not blow past the guess. + * + * They are xpsb_pds_dms_word()'s to compute, from the DDK's own + * arithmetic. The per-pixel cost it needs is the primary attribute + * footprint the iterators actually deliver, which is what + * xpsb_heap_set_attribs() has already put in the pixel size field - + * not the program's register count, which counts only the registers + * the code names and so misses everything a MOE repeat walks into. + * + * This is also the write that used to land on top of the one in + * xpsb_heap_set_attribs(): two places set the same field and the + * later, cruder one won. */ + { + const char *e = SGX_ENVS("SGX_PART_COUNT"); + unsigned pixel = *dms & XPSB_DMS_PIXELSIZE_MASK; + + /* n is a dword count and the calculation wants one: the + * vendor hands CalculatePixelDMSInfo() its + * ui32USESecAttribDataSizeInDwords. Bytes here counted four + * secondary chunks for every real one. */ + *dms = xpsb_pds_dms_word(*dms, pixel, fs ? fs->ntemps : 0u, n); + if (e && *e) + *dms = (*dms & ~0x3e000000u) | + ((strtoul(e, NULL, 0) << 25) & 0x3e000000u); + e = SGX_ENVS("SGX_TMP_COUNT"); + if (e && *e) + *dms = (*dms & ~0x00030000u) | + (((strtoul(e, NULL, 0) - 1u) << 16) & + 0x00030000u); + } + return 4; +} + +static int ctx_upload_pools(struct sgx_context *ctx) +{ + const struct sgx_shader *sh[2]; + unsigned i, base; + uint32_t *dst; + + if (!ctx->bound[SGX_CTX_BO_HEAP]) + return -EINVAL; + if (!ctx->bo[SGX_CTX_BO_HEAP].map) { + int ret = sgx_bo_map(ctx->ws, &ctx->bo[SGX_CTX_BO_HEAP]); + + if (ret) + return ret; + } + dst = ctx->bo[SGX_CTX_BO_HEAP].map; + + sh[0] = ctx->vs; + sh[1] = ctx->fs; + for (i = 0; i < 2; i++) { + uint64_t end; + + if (!sh[i] || !sh[i]->pool_dwords) + continue; + /* The fragment program's block mirrors the secondary attribute + * bank: its uniforms at offset zero, the sampler state blocks + * after them - left alone, the frame's own PDS writes those - + * and the literal pool at the offset the compiler chose. The + * vertex stage keeps the placement it already had. */ + base = (i == 1) ? SGX_FS_CONST_OFF / 4 : sh[i]->pool_base; + if (i == 1) + base += sh[i]->pool_base; + /* The vertex stage has no block of its own, so its pool_base + * is an sa register number and lands on the frame's own heap + * header - dword 0 is the render target's descriptor. The + * driver asks for no vertex literal pool, so this cannot + * happen; refuse rather than overwrite the frame silently if + * that ever changes. */ + else if (base * 4u < XPSB_PRI_PDS_OFF) + return -EINVAL; + end = (uint64_t)(base + sh[i]->pool_dwords) * 4; + if (end > ctx->bo[SGX_CTX_BO_HEAP].size) + return -ENOSPC; + memcpy(dst + base, sh[i]->pool, + sh[i]->pool_dwords * sizeof(uint32_t)); + if (SGX_ENV("SGX_DEBUG") && i == 1) { + unsigned q; + + fprintf(sgx_log(), "sgx: pool: %u dword(s) at sa%u:", + sh[i]->pool_dwords, sh[i]->pool_base); + for (q = 0; q < sh[i]->pool_dwords && q < 12; q++) + fprintf(sgx_log(), " %08x", sh[i]->pool[q]); + fputc('\n', sgx_log()); + } + ctx->uploads++; + } + return 0; +} + +/* sgxtri's apply_relocs(). A relocation names a source buffer whose GPU + * address the value is built from and a destination buffer the dword is + * written into; USE records go through the kernel, which owns the base + * registers. */ +#define SGX_RELOC_SHIFT_MASK 0x0000ffffu +#define SGX_RELOC_ALSHIFT_MASK 0xffff0000u +#define SGX_RELOC_ALSHIFT_SHIFT 16 + +static int ctx_apply_relocs(struct sgx_context *ctx, + const struct xpsb_reloc *rel, unsigned n, + const int *list, unsigned nlist, uint32_t **dstv, + const unsigned *dstn) +{ + unsigned i; + + for (i = 0; i < n; i++) { + const struct xpsb_reloc *r = &rel[i]; + uint32_t val, shift, align_shift, background; + uint32_t *dst; + uint64_t src; + + if (r->buffer >= nlist || r->dst_buffer >= nlist) + return -EINVAL; + src = ctx->bo[list[r->buffer]].gpu_va; + if (!src) + return -EINVAL; + if (!dstv[r->dst_buffer] || r->where >= dstn[r->dst_buffer]) + return -EINVAL; + dst = dstv[r->dst_buffer] + r->where; + background = r->background; + + if (r->reloc_op == XPSB_RELOC_OP_OFFSET) { + val = (uint32_t)src + r->pre_add; + } else if (r->reloc_op == XPSB_RELOC_OP_USE_REG || + r->reloc_op == XPSB_RELOC_OP_USE_OFFSET) { + uint32_t reg = 0, off = 0; + int ret = sgx_use_base(ctx->ws, src + r->pre_add, + r->arg0, r->arg1, ®, &off); + + if (ret) + return ret; + val = (r->reloc_op == XPSB_RELOC_OP_USE_REG) ? reg : off; + /* psb_sgx.c:758-766 - a USE_OFFSET record takes its + * background from the dword the USE_REG record before + * it just wrote, not from the table. */ + if (r->reloc_op == XPSB_RELOC_OP_USE_OFFSET) + background = *dst; + } else { + return -EINVAL; + } + + shift = r->shift & SGX_RELOC_SHIFT_MASK; + align_shift = (r->shift & SGX_RELOC_ALSHIFT_MASK) >> + SGX_RELOC_ALSHIFT_SHIFT; + val = (val >> align_shift) << shift; + *dst = (background & ~r->mask) | (val & r->mask); + } + return 0; +} + +/* Which context object stands in for each of the captured frame's buffers. + * The command buffer has no GPU address here - the streams are passed to the + * kernel by the submit - so it maps to the heap, which no relocation sources + * through it. */ +static const int ctx_ta_list[XPSB_FRAME_NBUF + 1] = { + SGX_CTX_BO_HEAP, SGX_CTX_BO_TARGET, SGX_CTX_BO_USSE, + SGX_CTX_BO_RASTGEOM, SGX_CTX_BO_HEAP, SGX_CTX_BO_VTX, + SGX_CTX_BO_DEPTH, SGX_CTX_BO_TEX, SGX_CTX_BO_TEX2 +}; + +/* xpsb_raster_list is { 0, -1, 2, 1, 3, 4 }; the -1 is the X-owned + * destination, which here is the render target. */ +static const int ctx_ras_list[XPSB_RAS_LIST_LEN] = { + SGX_CTX_BO_HEAP, SGX_CTX_BO_TARGET, SGX_CTX_BO_USSE, + SGX_CTX_BO_TARGET, SGX_CTX_BO_RASTGEOM, SGX_CTX_BO_HEAP +}; + +/* Copy the template user record once per group, patch the index range into + * each, and put the terminator after the last. The frame is generated with a + * single user record and the relocations have just filled in its addresses, so + * it is the template: what differs between records is the index count and + * where the indices start, and both are in the record. */ +/* The primary PDS data the pixel state names: twelve dwords at heap+0x340, + * with unit 0's texture control, format and address at the indices + * xpsb_pds_gen_primary() puts them at. A record after the first gets a copy so + * that the texture it samples is its own. */ +#define SGX_PDS_DATA_OFF XPSB_PRI_PDS_OFF +/* Where this frame's primary program actually is: a list too wide for the + * captured slot is built in the spare block, and the per-record copies have + * to be taken from wherever it landed. */ +/* Where a record's own program was built. Each record rebuilds the primary + * PDS from its own attributes, so a record whose list needs the spare block + * sits there while one that fits stays in the captured slot - and its copy has + * to be taken from whichever it is. Taking the frame's base for every record + * copied one record's data over another's program and stalled the core. */ +static unsigned ctx_rec_pri_pds_off(const struct sgx_context *ctx, unsigned k) +{ + if (k < ctx->nrange && ctx->range[k].fs && + !SGX_ENV("SGX_NO_PER_RECORD_ATTRIBS")) + return xpsb_pri_pds_off(&ctx->range[k].fs->attribs); + return ctx->pri_pds_off ? ctx->pri_pds_off : XPSB_PRI_PDS_OFF; +} + +static unsigned ctx_pri_pds_off(const struct sgx_context *ctx) +{ + /* What the builder actually chose - it may lay the frame out from + * attributes it synthesised rather than the program's own, so this + * is recorded there rather than derived again here. */ + return ctx->pri_pds_off ? ctx->pri_pds_off : XPSB_PRI_PDS_OFF; +} +#define SGX_PDS_DATA_DWORDS 12u +#define SGX_PDS_COPY_OFF (SGX_STATE_COPY_OFF + \ + SGX_MAX_DRAWS * SGX_STATE_WINDOW) +/* Where a group after the first puts its fragment program, and the three + * fields of the PDS data word that names it. The fields and their shifts are + * the relocation table's, for the record at heap dword 0xd0 with pre_add + * 0x100: the register in [3:0], then the offset twice, and each write after + * the first takes what the one before it left as its background. */ +/* Moved up from 0x10000 to leave room for the wider vertex program copies + * below it - see SGX_HWTCL_PROG_STRIDE. The object is two megabytes and the + * programs are packed from here, so the space this costs is not scarce. */ +#define SGX_USSE_GROUP_OFF 0x20000u +/* The records' programs are packed from SGX_USSE_GROUP_OFF, each taking only + * what it needs rounded to the sixteen bytes the USE word can name. A fixed + * kilobyte per record held a hundred and twenty-eight instructions and refused + * every longer program with -ENOSPC - gtk3-demo's compositing shaders are + * longer than that, so every frame of ninety draws or more was dropped and the + * window stayed empty. It also bounded a frame at 1984 records regardless of + * how short the programs were. */ +#define SGX_USSE_GROUP_ALIGN 16u +#define SGX_USE_REG_BG 0x180000u +#define SGX_USE_SIZE 0x8u +#define SGX_USE_DM 1u +/* The copy has to carry the code as well as the data: the descriptor names the + * data segment and the code begins at data + 4 * size, so a copy of the data + * alone leaves the program running whatever follows it - which faulted the + * vertex data master at address zero. */ +/* The step is the widest region a program can be built in, so a record's copy + * never runs into the next one's; the length copied is the region the frame's + * program actually occupies. */ +#define SGX_PDS_COPY_BYTES 0x80u +#define SGX_PDS_COPY_STEP 0x80u + +static unsigned ctx_rec_pri_pds_off(const struct sgx_context *ctx, unsigned k); + +static unsigned ctx_pri_pds_bytes(const struct sgx_context *ctx, unsigned k) +{ + return ctx_rec_pri_pds_off(ctx, k) == XPSB_PRI_PDS_OFF ? + SGX_PDS_COPY_BYTES : + XPSB_PRI_PDS_ALT_END - XPSB_PRI_PDS_ALT; +} + +/* The window of heap a draw's state block lives in, and where in it the record + * points. The captured user draw's block is at 0x4560 and the ISP word it uses + * is at 0x4520, so the window is taken from below the one to above the other. */ +#define SGX_STATE_WINDOW_OFF 0x4500u +#define SGX_STATE_WINDOW 0x100u +#define SGX_STATE_IN_WINDOW (0x4560u - SGX_STATE_WINDOW_OFF) +/* The per-record copy regions, laid out one after another from the record + * count rather than at round offsets. Each is indexed by record, so a region + * is SGX_MAX_DRAWS entries wide; spacing them 0x20000 apart put every one of + * them inside the next once a frame carried more than a few hundred records, + * and a record's state block was then overwritten by another record's + * constants. Derived so they cannot overlap, and checked below. */ +#define SGX_STATE_COPY_OFF 0x100000u /* well past the captured heap */ +/* A record with its own uniforms needs its own secondary program, because + * that program is what names the block the uniforms are read from; its + * binding is the first of the state block's PDS words. A record's copy of the + * window is addressed as if it were the heap - the same dwords, one window + * over - so the layout functions read the copy's own mask. */ +#define SGX_STATE_WINDOW_HEAP(heap, off) \ + ((heap) + ((off) - SGX_STATE_WINDOW_OFF) / 4u) +/* Per-record copies of the secondary program and of the uniform block it + * reads. Four dwords of program, and one block of constants each. */ +#define SGX_SEC_PDS_COPY_STEP 0x20u +/* How many records may take a constant copy in one frame. These two regions + * are handed out in order rather than indexed by record, because only a record + * with uniforms needs them: at one kilobyte a record the constant region alone + * was four megabytes of the five and a half the per-record regions have, which + * is what stopped SGX_MAX_DRAWS rising. A frame with more than this many such + * records splits, the same as any other budget. */ +#define SGX_MAX_CONST_COPIES 1024u +#define SGX_SEC_PDS_COPY_OFF (SGX_PDS_COPY_OFF + \ + SGX_MAX_DRAWS * SGX_PDS_COPY_STEP) +#define SGX_FS_CONST_COPY_STEP (SGX_MAX_FS_CONST_DW * 4u) +#define SGX_FS_CONST_COPY_OFF (SGX_SEC_PDS_COPY_OFF + \ + SGX_MAX_CONST_COPIES * SGX_SEC_PDS_COPY_STEP) +/* The end of the last per-record region: everything the heap keeps above the + * captured template has to fit below the object's size. */ +#define SGX_COPY_REGIONS_END (SGX_FS_CONST_COPY_OFF + \ + SGX_MAX_CONST_COPIES * SGX_FS_CONST_COPY_STEP) +_Static_assert(SGX_HWTCL_VPDS_COPY_OFF >= SGX_COPY_REGIONS_END, + "the vertex PDS copies sit inside a per-record copy region"); +_Static_assert(SGX_VPDS_REC_COPY_OFF >= SGX_HWTCL_VPDS_COPY_OFF + + SGX_HWTCL_MAX_UNI_SETS * SGX_HWTCL_VPDS_COPY_STRIDE, + "the per-record vertex PDS copies overlap the per-uniform ones"); +_Static_assert(SGX_VPDS_REC_COPY_OFF + + (uint64_t)SGX_MAX_DRAWS * SGX_HWTCL_VPDS_COPY_STRIDE <= 7864320u, + "the per-record vertex PDS copies do not fit the heap object"); +_Static_assert(SGX_HWTCL_VPDS_COPY_OFF + + SGX_HWTCL_MAX_UNI_SETS * SGX_HWTCL_VPDS_COPY_STRIDE <= 7864320u, + "the per-record copy regions do not fit the heap object"); + +/* The PDS data word that names a fragment program, built the way the frame's + * relocations build it. */ +static int ctx_use_word_for(struct sgx_context *ctx, uint32_t usse_off, + unsigned size, unsigned dm, uint32_t background, + uint32_t *out) +{ + uint32_t reg = 0, off = 0, w; + int ret = sgx_use_base(ctx->ws, + ctx->bo[SGX_CTX_BO_USSE].gpu_va + usse_off, + size, dm, ®, &off); + + if (ret) + return ret; + /* The word names the program by USE base register and offset: + * [3:0] the CR_USE_CODE_BASE register the kernel handed out, [7:4] the + * offset's bits 18:15 and [18:8] its bits 14:4 - one rotated 19-bit + * byte offset, sixteen-byte granular, inside that register's 512 KiB + * window. The high bits are the iterator, texture and DMA dependency + * flags. All three address fields are recomputed together, because + * they describe one address. */ + w = background; + w = (w & ~0x0000000fu) | (reg & 0x0000000fu); + w = (w & ~0x000000f0u) | (((off >> 0xf) << 4) & 0x000000f0u); + w = (w & ~0x0007ff00u) | (((off >> 4) << 8) & 0x0007ff00u); + *out = w; + return 0; +} + +static int ctx_use_word(struct sgx_context *ctx, uint32_t usse_off, uint32_t *out) +{ + return ctx_use_word_for(ctx, usse_off, SGX_USE_SIZE, SGX_USE_DM, + SGX_USE_REG_BG, out); +} + +/* The DOUTU at the state descriptor names the USE program that copies the + * block to the MTE, and that program is the block's size: one that copies + * fifteen hands a longer block over short. The frame's own is relocated + * from cfg.state_use_off; a record's copy is built here, on the vertex data + * master (the relocation's arg1), keeping the word's dependency bits. */ +static int ctx_state_use_word(struct sgx_context *ctx, uint32_t *win) +{ + unsigned n = xpsb_state_dwords(win[xpsb_heap_state_base(win)]); + uint32_t w; + int ret; + + if (!n || n > XPSB_USSE_STATE_MAX) + return -EINVAL; + ret = ctx_use_word_for(ctx, xpsb_usse_state_copy_off(n), + xpsb_usse_state_copy_size(n), 0u, + win[XPSB_HEAP_STATE_USE], &w); + if (ret) + return ret; + win[XPSB_HEAP_STATE_USE] = w; + return 0; +} + +/* Blending is the caller's own program with its last instruction replaced. + * + * The compositing shader's second instruction is the operator itself - + * "sop2.end o0, i0, o0, zero.comp, s1a.comp, add", which is o0 = src + o0 * + * (1 - src.alpha), reading the destination out of o0 - and its first source is + * whatever holds the source colour. In the X frame that is i0, the modulate of + * a source picture by a mask; here it is the register the caller's program + * wrote its result into, so the two fit together with one word changed. + * + * The register sits in bits [14:7] of the low word, the same field a move's + * source uses: assembling the instruction for r0, r1 and r16 gives 0x10000000, + * 0x10000080 and 0x10000800, and the compiled program's final "mov.end o0, rN" + * carries N in the same place. + */ +#define SGX_SRC_REG(w0) (((w0) >> 7) & 0xffu) +#define SGX_SRC_FIELD(r) (((uint32_t)(r) & 0xffu) << 7) + +/* Replace the program's closing move with the blend instructions. + * + * sgx_blend_build() makes them from the factors, given the register the + * caller's program wrote its result into - read out of the move - and the + * temporaries past the program's own for what a form has to stage: the + * constant, a pre-scaled destination, a masked result. The captured + * premultiplied-over operator is the fallback for a state the factors could + * not be read from. */ +/* Is this the closing "mov o0, rN" - the one instruction a blend may replace? + * Opcode group 0 with the output bank as destination and a temporary as the + * source; the pack instructions that write o0 directly are group 0x10. */ +static int sgx_insn_is_pixel_move(uint64_t insn) +{ + uint32_t w1 = (uint32_t)(insn >> 32); + + return ((w1 >> 27) & 0x1fu) == 0x05u; +} + +#define SGX_INSN_END_HI (1u << 18) +#define SGX_INSN_PRED_HI (7u << 24) + +/* Where the program writes its pixel. That is the last instruction only in a + * straight-line program: one that can discard ends "p1 mov o0, rN" with a + * nop.end behind it, so testing the last instruction alone refused every + * program with more than one exit - which is every glyph alacritty draws. */ +static int sgx_pixel_move_at(const struct sgx_shader *fs) +{ + unsigned last; + + if (!fs || !fs->ninsns) + return -1; + last = fs->ninsns - 1u; + if (sgx_insn_is_pixel_move(fs->code[last])) + return (int)last; + if (last && sgx_insn_is_pixel_move(fs->code[last - 1u])) + return (int)(last - 1u); + return -1; +} + +/* What replaces the move takes its predicate - a pixel the program discarded + * must not be blended into the target either - and leaves ending the program + * to the terminator, when the move had one behind it. */ +static uint64_t sgx_insn_take_pred(uint64_t insn, uint64_t from, unsigned tail) +{ + uint32_t w1 = (uint32_t)(insn >> 32); + + /* Not the predicate: hi[24] is the blend's own CMOD1, so [26:24] is + * not a predicate field in this group and writing one destroyed the + * equation - every glyph came out black. Where the group keeps its + * predicate is not established. */ + (void)from; + if (tail) + w1 &= ~SGX_INSN_END_HI; + return ((uint64_t)w1 << 32) | (insn & 0xffffffffu); +} + +static int ctx_blend_program(const struct sgx_shader *fs, enum sgx_blend_op op, + const struct sgx_blend_desc *d, uint32_t constant, + int have_insn, uint64_t *out, unsigned cap, + unsigned *n, unsigned *ntemps, unsigned logicop, + unsigned logicop_func) +{ + struct sgx_blend_desc over; + uint64_t ins[SGX_BLEND_MAX_INSNS]; + unsigned i, reg, tail, tmp, ln = 0, used = 0, q; + unsigned flags = SGX_ENV("SGX_BLEND_SPLIT") ? SGX_BLEND_F_SPLIT : 0u; + int pm, r; + + if (!have_insn && !logicop) { + if (op != SGX_BLEND_OVER) { + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: notsup: blend op %d " + "without an insn, %u insn(s)\n", + (int)op, fs ? fs->ninsns : 0); + return -ENOTSUP; + } + /* The one named operator: over, premultiplied. */ + memset(&over, 0, sizeof over); + over.rgb_src = over.alpha_src = 0x1; /* ONE */ + over.rgb_dst = over.alpha_dst = 0x13; /* INV_SRC_ALPHA */ + over.mask = 0xf; + over.enable = 1; + d = &over; + } + if (!fs || !fs->compiled || !fs->ninsns || fs->ninsns > cap) + return -EINVAL; + /* A state that names the second colour needs a program that wrote one. + * The two are settled separately - the blend state is bound and the + * program compiled without either knowing the other - so a mismatch is + * refused here rather than reading a register nothing filled. */ + if (sgx_blend_uses_src1(d) && !fs->has_src1) { + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: notsup: the blend wants a " + "second colour and the program wrote none\n"); + return -ENOTSUP; + } + /* Only a program whose last instruction is the move that writes the + * pixel can have that move replaced. + * + * A program that computes its colour ends with two pcku8f32 packing + * straight into o0, and replacing only the last of them left the first + * writing o0 one instruction before the blend reads it - so the + * destination the blend was meant to read was whatever the pack had + * just put there, and the register the blend was aimed at was a raw + * float temp. That is the shape the vendor never allows: turning + * blending on, its first act is to stop the program writing o0. + * Refusing here degrades to an unblended draw, which is visibly wrong + * rather than quietly corrupt. */ + pm = sgx_pixel_move_at(fs); + if (pm < 0) { + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: notsup: no pixel move: %u " + "insn(s), last %016llx\n", fs->ninsns, + (unsigned long long)fs->code[fs->ninsns - 1]); + return -ENOTSUP; + } + tail = fs->ninsns - 1u - (unsigned)pm; + for (i = 0; i < fs->ninsns; i++) + out[i] = fs->code[i]; + reg = SGX_SRC_REG((uint32_t)(out[pm] & 0xffffffffu)); + tmp = fs->ntemps; + + /* The blend reads its source from a register, and the field it is + * given carries a number without a bank - so it is only the right + * register when the move it replaced took its source from a + * temporary. A program that hands the iterated colour straight to the + * pixel ends "mov o0, pa0", and the blend then read r0: whatever + * happened to be there, which for a passthrough program is nothing. + * That is a blended draw come out black. + * + * So the move is kept, aimed at a temporary one past the program's + * own, and the blend reads that. The caller widens the task's + * temporary count to cover it. */ + { + uint32_t w0 = (uint32_t)(out[pm] & 0xffffffffu); + uint32_t w1 = (uint32_t)(out[pm] >> 32); + unsigned bank = ((w0 >> 30) & 3u) | (((w1 >> 17) & 1u) << 2); + + if (bank != 0u) { /* not the temporary bank */ + if (fs->ntemps + 1u > SGX_TEMP_FIELD_MAX) + return -ENOSPC; + reg = fs->ntemps; + tmp = reg + 1u; + /* the same move, into that temporary: destination + * bank becomes 0 and its register number `reg` */ + w1 &= ~0x00080003u; + w0 = (w0 & ~0x0fe00000u) | ((reg & 0x7fu) << 21); + out[pm] = ((uint64_t)w1 << 32) | w0; + if (logicop) { + if (sgx_logicop_insn(logicop_func, reg, ins, + &ln)) + return -ENOTSUP; + } else { + r = sgx_blend_build(d, constant, reg, + fs->src1_reg, tmp, + SGX_TEMP_FIELD_MAX - tmp, + flags, ins, + SGX_BLEND_MAX_INSNS, &ln, + &used); + if (r) + return r; + } + if (fs->ninsns + ln > cap) + return -ENOSPC; + /* Behind the move, ahead of the terminator that the + * program's other exits reach. */ + memmove(&out[(unsigned)pm + 1u + ln], + &out[(unsigned)pm + 1u], + (size_t)tail * sizeof *out); + for (q = 0; q < ln; q++) + out[(unsigned)pm + 1u + q] = + sgx_insn_take_pred(ins[q], out[pm], + tail); + *n = fs->ninsns + ln; + /* The registers just written are the highest the + * code touches, so the program needs one past them. */ + *ntemps = tmp + used; + return 0; + } + } + if (logicop) { + /* The move that wrote the pixel is replaced: the logic op + * writes o0 itself, out of the register that move read. */ + if (fs->ninsns + 1u > cap || + sgx_logicop_insn(logicop_func, reg, ins, &ln)) + return -ENOTSUP; + if (fs->ninsns - 1u + ln + tail > cap) + return -ENOTSUP; + memmove(&out[(unsigned)pm + ln], &out[(unsigned)pm + 1u], + (size_t)tail * sizeof *out); + for (q = 0; q < ln; q++) + out[(unsigned)pm + q] = + sgx_insn_take_pred(ins[q], fs->code[pm], tail); + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: logicop %u: %u insn(s) from " + "reg %u, first %016llx\n", logicop_func, ln, + reg, (unsigned long long)out[pm]); + *n = fs->ninsns - 1u + ln; + *ntemps = fs->ntemps; + return 0; + } + r = sgx_blend_build(d, constant, reg, fs->src1_reg, tmp, + SGX_TEMP_FIELD_MAX - tmp, + flags, ins, SGX_BLEND_MAX_INSNS, &ln, &used); + if (r) + return r; + if (fs->ninsns - 1u + ln > cap) + return -ENOSPC; + /* A program that can discard ends on a *predicated* move, and the blend + * cannot take that predicate: hi[24] is its own CMOD1, so this group + * keeps no predicate where every other one does. + * + * Three ways were measured. Replacing the move outright blends the + * fragments the shader discarded. Aiming the blend at a temporary + * fails because SOP2 reads its own destination as the "dst" its + * factors select, so it reads an uninitialised register. Saving o0 and + * putting it back for the discarded fragments fails too - o0 is not + * readable as a source, so the restore wrote a constant over every + * discarded fragment, which erased alacritty's glyphs the moment its + * background pass started discarding as it should. + * + * What is left is to leave the blend unpredicated and make it a no-op + * for the fragments that did not survive: with a source of zero, a + * destination factor of ONE, 1 - src or 1 - src.a is one, and the + * pixel keeps what it found. So the source is zeroed under the inverse + * predicate, and a blend whose destination factor is anything else is + * refused rather than drawn wrongly. Only a single SOP2 is analysed + * that way; a list (a constant, a mask) is refused. */ + if (tail && ((uint32_t)(fs->code[pm] >> 32) >> 24 & 7u)) { + static const unsigned char inv[5] = { 0, 5, 6, 0, 0 }; + unsigned pred = (uint32_t)(fs->code[pm] >> 32) >> 24 & 7u; + uint32_t hi = (uint32_t)(ins[0] >> 32), lo = (uint32_t)ins[0]; + unsigned csel2 = (hi >> 3) & 7u, asel2 = (hi >> 9) & 3u; + uint32_t w0, w1; + + if (ln != 1u || ((hi >> 27) & 0x1fu) != 0x10u) + return -ENOTSUP; + if (pred > 2u || !inv[pred]) + return -ENOTSUP; /* no inverse for p2 and p3 */ + /* The zeroed source is operand 1 below; under the swap it + * is operand 2 and the factor test is the other field's. */ + if (SGX_BLEND_LO_SWAPPED(lo)) + return -ENOTSUP; + if (!(hi & (1u << 15)) || /* CMOD2, the invert */ + (csel2 != 0u && csel2 != 1u && csel2 != 3u) || + !(hi & (1u << 2)) || /* AMOD2 */ + (asel2 != 0u && asel2 != 1u)) + return -ENOTSUP; + if (fs->ninsns + 1u > cap) + return -ENOSPC; + memmove(&out[(unsigned)pm + 1u], &out[pm], + (size_t)(tail + 1u) * sizeof *out); + + /* !P mov rS, c48 - the zero this part keeps in the constant + * bank, bank 4 register 48. */ + w0 = (uint32_t)(fs->code[pm] & 0xffffffffu); + w1 = (uint32_t)(fs->code[pm] >> 32); + w1 &= ~(0x00080003u | SGX_INSN_PRED_HI | SGX_INSN_END_HI); + w1 |= 0x00020000u; /* source bank bit 2 */ + w1 |= (uint32_t)inv[pred] << 24; + w0 = (w0 & ~0x0fe00000u) | ((reg & 0x7fu) << 21); + w0 = (w0 & ~0x00007f80u & ~0xc0000000u) | SGX_SRC_FIELD(48u); + out[pm] = ((uint64_t)w1 << 32) | w0; + + /* and the blend, in the move's place, writing o0 */ + out[(unsigned)pm + 1u] = + sgx_insn_take_pred(ins[0], fs->code[pm], 1u); + *n = fs->ninsns + 1u; + *ntemps = fs->ntemps; + return 0; + } + /* The list in the move's place; the last of it ends the program + * unless the terminator behind the move does. */ + memmove(&out[(unsigned)pm + ln], &out[(unsigned)pm + 1u], + (size_t)tail * sizeof *out); + for (q = 0; q < ln; q++) + out[(unsigned)pm + q] = + sgx_insn_take_pred(ins[q], fs->code[pm], tail); + *n = fs->ninsns - 1u + ln; + /* The blend reads the register the move already read, plus whatever + * the list staged past the program's own temporaries. */ + *ntemps = tmp + used; + return 0; +} + +/* Put one group's fragment program in the slot it will be named at. */ +static int ctx_put_program(struct sgx_context *ctx, unsigned k, uint32_t off, + unsigned *ntemps, uint32_t *bytes) +{ + uint32_t *slot; + unsigned i; + + { + /* Heap rather than stack, for the same reason as above. */ + uint64_t *code = NULL; + const uint64_t *src = ctx->range[k].fs ? ctx->range[k].fs->code : NULL; + unsigned n = ctx->range[k].fs ? ctx->range[k].fs->ninsns : 0; + unsigned nt = ctx->range[k].fs ? ctx->range[k].fs->ntemps : 0; + uint32_t need; + int rc = 0; + + /* Gated on whether the record blends, not on whether its + * factors happened to match a named operator. sgx_blend_insn() + * builds the instruction for any ordinary factor pair, and + * testing the enum threw all of those away - SRC_ALPHA with + * ONE_MINUS_SRC_ALPHA, the commonest blend in GL, was simply + * drawn unblended. */ + if (ctx->range[k].blended) { + if (!n) + return -EINVAL; + code = malloc((size_t)(n + 1u + SGX_BLEND_MAX_INSNS) * + sizeof *code); + if (!code) + return -ENOMEM; + rc = ctx_blend_program(ctx->range[k].fs, + ctx->range[k].blend, + &ctx->range[k].blend_desc, + ctx->range[k].blend_const, + ctx->range[k].blend_insn, + code, n + 1u + SGX_BLEND_MAX_INSNS, + &n, &nt, + ctx->range[k].logicop, + ctx->range[k].logicop_func); + if (rc) { + /* Only one operator is built this way so far. + * An operator that is not draws unblended - + * wrong, and visibly so - rather than failing + * the frame, which would drop the geometry + * entirely and show nothing at all. */ + free(code); + code = NULL; + n = ctx->range[k].fs->ninsns; + nt = ctx->range[k].fs->ntemps; + src = ctx->range[k].fs->code; + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: blend operator %d " + "is not built; drawing " + "unblended\n", + (int)ctx->range[k].blend); + } else { + src = code; + } + } + if (!src || !n) { + free(code); + return -EINVAL; + } + /* Only now is the length known: a blend appends an + * instruction, and an operator that will not build falls back + * to the caller's own program. */ + need = ((uint32_t)n * 8u + SGX_USSE_GROUP_ALIGN - 1u) & + ~(SGX_USSE_GROUP_ALIGN - 1u); + if (off + need > ctx->bo[SGX_CTX_BO_USSE].size) { + free(code); + return -ENOSPC; + } + slot = (uint32_t *)((char *)ctx->bo[SGX_CTX_BO_USSE].map + off); + memset(slot, 0, need); + for (i = 0; i < n; i++) { + slot[i * 2] = (uint32_t)(src[i] & 0xffffffffu); + slot[i * 2 + 1] = (uint32_t)(src[i] >> 32); + } + /* What the code written into the slot references, not the + * program's count with one added for a blend that may not + * have appended anything. */ + *ntemps = nt; + *bytes = need; + free(code); + } + return 0; +} + +/* Whether the bound fragment program samples for itself, so its sampler + * descriptors live in the constant block behind its uniforms. */ +/* A program that samples for itself keeps its sampler descriptors in the + * constant block behind its uniforms; one fed by the iterator has none. */ +static int fs_samples(const struct sgx_shader *fs) +{ + return fs && !fs->tex_preiterated; +} + +/* Whether record k's program fits the two things the frame decided for it. + * + * ctx_sec_pds() sizes one transfer from record 0's program and every record + * copies that program verbatim; the same word carries the USE partition + * split, sized from record 0's register demand. Both are silent when they are + * too small: the first leaves the top of the bank holding whatever the last + * program left there, the second gives a task less of the register file than + * its program names. Counted rather than printed per record, because a frame + * carries hundreds. */ +static unsigned nfit_said; + +/* Where a record's own program expects the frame to have put its attributes, + * against where the frame actually put them. The iterator is configured once + * per frame from record 0's program, while each record's code was compiled + * against its own - so a record whose program lays its issues out differently + * reads its colour, or a coordinate set, at the wrong register. */ +static int ctx_record_bases_differ(const struct sgx_shader *fs, + const struct sgx_shader *r0, + unsigned *mine, unsigned *frames) +{ + unsigned c0 = 0, c1 = 0, sb0[XPSB_NSET_MAX], sb1[XPSB_NSET_MAX]; + unsigned tb0[XPSB_MAX_TEX], tb1[XPSB_MAX_TEX], q; + + if (!fs || !r0 || fs == r0) + return 0; + memset(sb0, 0, sizeof sb0); memset(sb1, 0, sizeof sb1); + memset(tb0, 0, sizeof tb0); memset(tb1, 0, sizeof tb1); + xpsb_attrib_bases(&fs->attribs, &c0, sb0, tb0); + xpsb_attrib_bases(&r0->attribs, &c1, sb1, tb1); + *mine = c0; + *frames = c1; + if (fs->attribs.colour && r0->attribs.colour && c0 != c1) + return 1; + for (q = 0; q < fs->attribs.nset && q < XPSB_NSET_MAX; q++) + if (q < r0->attribs.nset && sb0[q] != sb1[q]) { + *mine = sb0[q]; + *frames = sb1[q]; + return 1; + } + return 0; +} + +static void ctx_check_record_fits(const struct sgx_context *ctx, unsigned k, + unsigned *nsa, unsigned *nsplit) +{ + const struct sgx_shader *fs = ctx->range[k].fs ? ctx->range[k].fs : + ctx->fs; + const uint32_t *heap = ctx->bo[SGX_CTX_BO_HEAP].map; + unsigned split, want; + + if (!fs) + return; + /* Zero included: a frame whose first program has no constants keeps + * the captured bare HALT, and then no record's constants are loaded + * at all. */ + if (fs->pool_base + fs->pool_dwords > ctx->sec_sa_dwords) + (*nsa)++; + if (!heap) + return; + split = (heap[xpsb_heap_pds_dw(heap, XPSB_PDS_W_CTL)] >> 25) & 0x1fu; + want = fs->ntemps + fs->nprimattr; + if (split && want && want > SGX_USE_REG_POOL / split) + (*nsplit)++; +} + +static int ctx_emit_draw_records(struct sgx_context *ctx) +{ + uint32_t pri_mir[XPSB_PRI_PDS_ALT_END / 4 - XPSB_PRI_PDS_ALT / 4]; + unsigned pri_mir_off = ~0u, pri_mir_len = 0; + uint32_t win_mir[SGX_STATE_WINDOW / 4]; + int win_mir_ok = 0; + + uint32_t *v = ctx->bo[SGX_CTX_BO_VTX].map; + unsigned first = 0u; + unsigned at = XPSB_USER_DRAW - first; /* the template's index */ + uint32_t tmpl[XPSB_DRAW_STRIDE], term[XPSB_DRAW_STRIDE]; + uint32_t swin[SGX_STATE_WINDOW / 4u]; + uint32_t ssec[SGX_SEC_PDS_COPY_STEP / 4u]; + uint32_t scon[SGX_FS_CONST_COPY_STEP / 4u]; + uint32_t idx_base; + uint32_t usse_next; + unsigned k, nrec = 0; + unsigned nsa_short = 0, nsplit_short = 0, nbase_off = 0; + unsigned base_mine = 0, base_frame = 0; + int only, upto, skip; + + /* One record still needs writing when it is bound to a copy of the + * vertex PDS program rather than to the frame's own. */ + if (!v || !ctx->nrange) + return 0; + /* Dword 4 of the user record is the DDK's index list word 2, and + * [26:25] of it is the vertex data master's own flat-shade selector. + * It has to name the corner the record's MTE shade model does - the + * vendor writes the two from one model + * (opengles1/validate.c:5013-5025) - and it stayed at the template's + * VERTEX0 for every draw this driver has ever made. Written on the + * frame's own record here so the single-record shortcut below carries + * it as well as the per-record path. */ + { + uint32_t *iw = (uint32_t *)((char *)v + ctx->draw_cmd_off) + 2; + + *iw = xpsb_vdm_idx_word(*iw, + sgx_raster_flat_corner(ctx->range[0].cull)); + } + /* One record can still need the treatment: the frame's own state block + * names its texture through the captured relocation, which points at + * the context's own buffer. A record whose texture is somewhere else + * has to carry its own copy or it samples that empty buffer - which + * is a black texture, with everything else about the draw correct. */ + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), + "sgx: records: nrange %u, hwtcl %d, nlimm %u, tex %d va 0x%x/0x%x, tex1 %d va 0x%x/0x%x\n", + ctx->nrange, ctx->hwtcl, + ctx->vs ? ctx->vs->nlimm : 0, + ctx->nrange ? ctx->range[0].tex[0].bound : 0, + ctx->nrange ? ctx->range[0].tex[0].va : 0, + (uint32_t)ctx->bo[SGX_CTX_BO_TEX].gpu_va, + ctx->nrange ? ctx->range[0].tex[1].bound : 0, + ctx->nrange ? ctx->range[0].tex[1].va : 0, + (uint32_t)ctx->bo[SGX_CTX_BO_TEX2].gpu_va); + /* SGX_FORCE_PER_RECORD takes the per-record path even when the frame + * could serve the one record itself, which separates that path from + * the addressing change that normally selects it. */ + /* And only when this record actually samples the textures the frame's + * template was built around. A texture stays bound after the draw that + * used it, so a solid record following a textured one has a unit bound + * at the frame's own address while sampling nothing - the va check + * above passes, but the template's state block and primary PDS still + * describe the textured draw, and the shortcut hands the solid record + * that state. That is twm's menu: three records paint it, then its + * border alone in the next frame takes the shortcut and runs the + * fill's state over its own geometry, half the pixmap in the wrong + * colour. When a texture is bound but the record's program reads no + * sampled set, take the per-record path, which rebuilds both. */ + if (!SGX_ENV("SGX_FORCE_PER_RECORD") && + ctx->nrange < 2 && !(ctx->hwtcl && ctx->vs && ctx->vs->nlimm) && + !(ctx->nrange && ctx->range[0].tex[0].bound && + ctx->range[0].tex[0].va != (uint32_t)ctx->bo[SGX_CTX_BO_TEX].gpu_va) && + !(ctx->nrange && ctx->range[0].tex[1].bound && + ctx->range[0].tex[1].va != (uint32_t)ctx->bo[SGX_CTX_BO_TEX2].gpu_va) && + !(ctx->nrange && ctx->range[0].fs && + ctx->range[0].fs->attribs.nset == 0 && + (ctx->range[0].tex[0].bound || ctx->range[0].tex[1].bound))) + return 0; /* the frame already holds the one */ + if ((ctx_ta_prefix(ctx) + (at + ctx->nrange + 1) * XPSB_DRAW_STRIDE) * 4 > + XPSB_VTX_OFF) + return -ENOSPC; + + v += ctx_ta_prefix(ctx); + /* Record zero keeps the frame's own state block, so its uniforms are + * the frame's block - which ctx_refresh_fs_constants() has been + * overwriting with whatever was bound last. Put the set this record + * actually opened with back, or the first record of a frame renders + * with the last draw's values. */ + { + uint32_t *heap = ctx->bo[SGX_CTX_BO_HEAP].map; + unsigned set = ctx->range[0].fs_uni; + + const struct sgx_shader *r0 = ctx->range[0].fs ? + ctx->range[0].fs : ctx->fs; + unsigned nu = ctx->fs_uni_set_dwords ? + ctx->fs_uni_set_dwords[set] : 0; + + /* Bounded like ctx_refresh_fs_constants() bounds its own copy. + * Unbounded, a set wider than the program's uniforms wrote + * over the sampler descriptors that sit behind them, and a + * record reading the frame's block sampled at the float it + * found there - the render faulted at 0x3f800000 with the + * texture cache as the requestor. */ + if (r0 && nu > r0->pool_base) + nu = r0->pool_base; + if (r0 && !r0->tex_preiterated) { + unsigned lim = fs_smp_base(r0, ctx_smp_units(ctx, r0)); + + if (nu > lim) + nu = lim; + } + if (heap && ctx->fs_uni_set && set < ctx->nfs_uni_set && nu && + SGX_FS_CONST_OFF + nu * 4u <= + ctx->bo[SGX_CTX_BO_HEAP].size) + ctx_pack_uniforms(r0, ctx->fs_uni_set[set], + ctx->fs_uni_set_dwords[set], + heap + SGX_FS_CONST_OFF / 4, nu); + /* And the descriptors back, in case the copy reached them. */ + if (heap && r0 && !r0->tex_preiterated) + ctx_write_sampler_slots(ctx, r0, + ctx->nrange ? &ctx->range[0] : + NULL, + heap + SGX_FS_CONST_OFF / 4, + SGX_FS_CONST_OFF); + } + memcpy(tmpl, v + at * XPSB_DRAW_STRIDE, sizeof tmpl); + memcpy(term, v + (at + 1) * XPSB_DRAW_STRIDE, sizeof term); + /* A runaway stream stops the core rather than failing the frame, so a + * terminator that is not one refuses the frame instead. */ + if (term[2] != XPSB_DRAW_TERM_CMD) + return -EINVAL; + idx_base = tmpl[3]; + usse_next = SGX_USSE_GROUP_OFF; + upto = -1; + skip = -1; + + /* SGX_ONLY_RECORD=n emits just that one record of the frame. A record + * that renders alone but not beside another says the defect is in + * carrying several primitive blocks, not in the record itself. */ + { + const char *e = SGX_ENVS("SGX_ONLY_RECORD"); + + only = (e && *e) ? (int)strtol(e, NULL, 0) : -1; + /* SGX_MAX_RECORD=n keeps records 0..n, so a defect that needs + * several records together can be bisected where SGX_ONLY_- + * RECORD, which keeps one, shows nothing. */ + e = SGX_ENVS("SGX_MAX_RECORD"); + upto = (e && *e) ? (int)strtol(e, NULL, 0) : -1; + /* And SGX_SKIP_RECORD=n drops just that one, which is what + * bisects a defect that only appears once several records are + * drawn together. */ + e = SGX_ENVS("SGX_SKIP_RECORD"); + skip = (e && *e) ? (int)strtol(e, NULL, 0) : -1; + } + /* Every record copies the frame's state window. Reading it out of the + * heap once leaves the per-record copy a write: the heap is + * write-combined, and reads from it are uncached. */ + memcpy(swin, (const char *)ctx->bo[SGX_CTX_BO_HEAP].map + + SGX_STATE_WINDOW_OFF, SGX_STATE_WINDOW); + memcpy(ssec, (const char *)ctx->bo[SGX_CTX_BO_HEAP].map + + XPSB_SEC_PDS_OFF, SGX_SEC_PDS_COPY_STEP); + memcpy(scon, (const char *)ctx->bo[SGX_CTX_BO_HEAP].map + + SGX_FS_CONST_OFF, SGX_FS_CONST_COPY_STEP); + + for (k = 0; k < ctx->nrange; k++) { + uint32_t *r = v + (at + nrec) * XPSB_DRAW_STRIDE; + /* Dwords in this record's own primary PDS program, when it + * carries one rather than the frame's. */ + unsigned rec_pri_dwords = 0; + + if (upto >= 0 && (int)k > upto) + continue; + if (skip >= 0 && (int)k == skip) + continue; + if (only >= 0 && (int)k != only) + continue; + nrec++; + /* Two of the fields a record copies are built once per frame + * from record 0's program: the secondary program's transfer + * size and the USE partition split. A record whose own + * program reaches further than either gets no diagnostic + * otherwise - its constants come from whatever the bank held, + * or its task cannot fit its share of the register file. */ + ctx_check_record_fits(ctx, k, &nsa_short, &nsplit_short); + if (ctx_record_bases_differ(ctx->range[k].fs, + sgx_frame_fs(ctx), &base_mine, + &base_frame)) + nbase_off++; + memcpy(r, tmpl, sizeof tmpl); + r[2] = sgx_ctx_draw_cmd_for(ctx, ctx->range[k].objtype, + ctx->range[k].count); + r[3] = idx_base + ctx->range[k].first * 2u; + /* Word 6 counts the vertex record in four-dword granules. The + * template carries the count the frame was built with, which + * is fixed before any fragment program is bound, while the + * record's real width is settled at flush - so a program + * whose attributes widened the record left every draw + * describing the narrower one. */ + r[6] = XPSB_DRAW_KIND((ctx->vtx_floats + 3u) / 4u); + /* And this record's own corner, which a frame of several + * rasterizer states does not share. */ + r[XPSB_DRAW_IDX_WORD] = xpsb_vdm_idx_word( + tmpl[XPSB_DRAW_IDX_WORD], + sgx_raster_flat_corner(ctx->range[k].cull)); + + /* SGX_VTX_COPY0 binds every record, the first included, to a + * copy of the vertex PDS program rather than to the frame's + * own. On a frame with one record that is the narrowest test + * of whether a copy runs at all. */ + /* SGX_NO_VPDS_COPY leaves the record on the frame's own vertex + * PDS program, which tells a defect in the per-record copy + * from one in the uniforms it was made to carry. */ + /* Its own copy, and with it its own vertex width: the fetch's + * DMA control and byte stride are that program's data, so + * records of different widths no longer have to be separate + * frames. Under hardware transform the copy also carries the + * uniform set's execution address, which is why it is keyed + * by record and not by either one alone. */ + if (!SGX_ENV("SGX_NO_VPDS_COPY")) { + uint32_t *heap = ctx->bo[SGX_CTX_BO_HEAP].map; + uint64_t end = (uint64_t)SGX_VPDS_REC_COPY_OFF + + (uint64_t)(k + 1) * SGX_HWTCL_VPDS_COPY_STRIDE; + unsigned slot = ctx->range[k].vpds_prog ? + ctx->range[k].uni : SGX_VPDS_NO_PROG; + + if (!heap || end > ctx->bo[SGX_CTX_BO_HEAP].size) + return -EINVAL; + r[5] = sgx_vpds_rec_copy(heap, + sgx_ctx_va[SGX_CTX_BO_HEAP], + sgx_ctx_va[SGX_CTX_BO_USSE], + k, slot, + ctx->range[k].vtx_floats); + } + + /* Every record gets a copy of the state block, so that what it + * says about itself is its own. The block is copied whole, + * window and all, and the record points at the same place + * inside the copy that it pointed at inside the original. + * + * The first record used to keep the frame's own block, and + * that block names its texture through the captured frame's + * relocation - which is why every texture had to be moved to + * one fixed address and the frame ended whenever a draw + * wanted a different one. The vendor has no such address: + * each record's PDS data carries the texture's own base, and + * a scene holds as many textures as the parameter heap has + * room for (gl-re/textures.md, work on the PDS primary + * program). Giving record zero the same copy as the rest is + * what lets us do the same. */ + { + uint32_t *heap = ctx->bo[SGX_CTX_BO_HEAP].map; + uint32_t off = SGX_STATE_COPY_OFF + k * SGX_STATE_WINDOW; + uint64_t va = ctx->bo[SGX_CTX_BO_HEAP].gpu_va; + uint32_t *win; + + if (!heap || off + SGX_STATE_WINDOW > + ctx->bo[SGX_CTX_BO_HEAP].size) + return -ENOSPC; + /* This record's own attribute configuration, put in + * place before the window and the PDS data below are + * copied out of it. The frame installs one, from + * whatever program happened to be bound when it was + * flushed - so a record whose program describes its + * varyings differently read registers the iterator + * never filled, which is a draw that renders flat. + * The group-14 word, the group-10 width and the + * primary program's pointer all live in the window, + * so each record can carry its own. + * + * The record width does not: it is three fields + * outside the window and one frame names one width, + * which is what set_record_width puts back. Draws + * that need different widths are kept out of one + * frame by the stride check in sgx_draw(). */ + if (SGX_ENV("SGX_DEBUG") && !ctx->range[k].fs) + fprintf(sgx_log(), "sgx: record %u: no program, " + "so it keeps the frame's attribs\n", k); + if (ctx->range[k].fs && + !SGX_ENV("SGX_NO_PER_RECORD_ATTRIBS")) { + unsigned dw = 0; + + if (!xpsb_heap_set_attribs(heap, + &ctx->range[k].fs->attribs, + &dw)) { + rec_pri_dwords = dw; + /* The record's width, not the + * frame's: group 10 is written into + * the template and copied into this + * record's own window below, and the + * fetch's DMA control and stride ride + * in the record's vertex PDS copy, + * which sgx_vpds_rec_copy() already + * patches with range[k].vtx_floats. */ + xpsb_heap_set_record_width(heap, + ctx->range[k].vtx_floats ? + ctx->range[k].vtx_floats : + ctx->vtx_floats, 0); + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: record " + "%u: own attribs, %u " + "set(s), pri %u dw\n", + k, ctx->range[k].fs-> + attribs.nset, dw); + } else if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: record %u: " + "%u set(s) REFUSED\n", k, + ctx->range[k].fs->attribs.nset); + } + /* The frame's window is the same 256 bytes for every + * record, and the heap is write-combining: reading it + * back is uncached, and this read it once per record. + * Taken once into cached memory instead - nothing in + * the loop writes the template, only its copies. */ + if (!win_mir_ok) { + memcpy(win_mir, (const char *)heap + + SGX_STATE_WINDOW_OFF, + SGX_STATE_WINDOW); + win_mir_ok = 1; + } + memcpy((char *)heap + off, win_mir, SGX_STATE_WINDOW); + /* This record's own ISP words, laid out in the copy - + * which may move its block, so the copy's descriptor + * is pointed at the block afterwards rather than + * inheriting the original's. */ + { + struct sgx_isp_set ff, bf; + int ret; + + win = SGX_STATE_WINDOW_HEAP(heap, off); + /* Group 8 is the record's: the draw module + * already applied the viewport and hands over + * window coordinates, so a frame holding both + * kinds cannot answer for both at once. */ + ctx_mte_viewport_into(ctx, win, + ctx->range[k].hwtcl); + ctx_isp_for_record(ctx, k, &ff, &bf); + ret = xpsb_heap_set_isp(win, &ff.a, &bf.a); + if (ret) + return ret; + /* And its own cull word, per record like the + * ISP words. */ + if (!SGX_ENV("SGX_NO_CULL")) { + ret = xpsb_heap_state_set(win, + XPSB_STATE_CULL, + ctx->range[k].cull_on, + &ctx->range[k].cull); + if (ret) + return ret; + } + win[XPSB_HEAP_STATE_DESC] = + (uint32_t)(va + off - SGX_STATE_WINDOW_OFF + + xpsb_heap_state_base(win) * 4u); + ret = ctx_state_use_word(ctx, win); + if (ret) + return ret; + /* The state task's primary attributes, word 1 + * [7:0] in blocks of four: the template's four + * covered the captured fifteen and no more. */ + r[1] = (r[1] & ~0xffu) | + (xpsb_heap_state_regs(win) & 0xffu); + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: record %u: state " + "block at %04x (heap +0x%x), mask " + "%08x, %u dwords, desc %08x ctl " + "%08x, doutu %08x, regs %u, cull %08x/%d\n", k, + xpsb_heap_state_base(win), + off - SGX_STATE_WINDOW_OFF + + xpsb_heap_state_base(win) * 4u, + win[xpsb_heap_state_base(win)], + xpsb_state_dwords( + win[xpsb_heap_state_base(win)]), + win[XPSB_HEAP_STATE_DESC], + win[XPSB_HEAP_STATE_DESC + 1], + win[XPSB_HEAP_STATE_USE], + r[1] & 0xffu, ctx->range[k].cull, + ctx->range[k].cull_on); + } + + /* And its own copy of the PDS data, so the texture the + * unit samples and the program it runs are this + * record's rather than the frame's. */ + { + uint32_t pds = SGX_PDS_COPY_OFF + + k * SGX_PDS_COPY_STEP; + uint32_t *d = (uint32_t *)((char *)heap + pds); + uint32_t *w; + unsigned t; + + if (pds + SGX_PDS_COPY_STEP > + ctx->bo[SGX_CTX_BO_HEAP].size) + return -ENOSPC; + { + unsigned so = ctx_rec_pri_pds_off(ctx, k); + unsigned sl = ctx_pri_pds_bytes(ctx, k); + + /* The block is read out of the heap, + * which is write-combining, so the + * read is uncached - and it ran once + * per record. A frame has at most two + * of these blocks, so each is taken + * once into cached memory and every + * record copies from there. */ + if (sl <= sizeof pri_mir) { + if (so != pri_mir_off || + sl != pri_mir_len) { + memcpy(pri_mir, + (char *)heap + so, + sl); + pri_mir_off = so; + pri_mir_len = sl; + } + memcpy(d, pri_mir, sl); + } else { + memcpy(d, (char *)heap + so, sl); + } + } + /* Every unit the record samples, rather than + * unit 0 and then unit 1 spelled out again: + * the PDS data carries a descriptor triple per + * issue, and which dwords a unit's triple + * lands on comes from the record's own + * program. Two units was never a hardware + * limit - the part has eight and the vendor + * compiler fills a format record for each - + * and every place that named the second one + * by hand was a place the third had to be + * added to. */ + /* Every texture issue of the record's own + * program gets a mapped address before the + * bound units overwrite theirs: the copy was + * taken before the frame's relocations ran, so + * an issue no unit patches - the dummy a + * textureless program carries so the pixel + * pipeline batches - would fetch the captured + * placeholder, which is unmapped, and the + * render faulted on it with the cache as the + * requestor. The context's own texture is + * always bound and nothing reads the texel. */ + { + unsigned char iof[XPSB_MAX_TEX]; + unsigned ni = 0, nu, u; + struct sgx_sampler_state dst; + struct sgx_sampler_view dv; + uint32_t dw0 = 0, dw1 = 0; + int dok; + + /* Described as well as addressed. A + * mapped address alone left the issue's + * own control words naming whatever + * extent the record was built with, and + * the fetch then ran off the end of + * this object: the cache faulted at + * 0x82555000, more than a megabyte past + * a buffer of 256 KiB, on every + * glmark2 terrain frame. A 32x1 + * texture fits inside whatever is + * mapped and nothing reads the texel. */ + memset(&dv, 0, sizeof dv); + memset(&dst, 0, sizeof dst); + dv.gpu_va = (uint32_t) + sgx_ctx_va[SGX_CTX_BO_TEX]; + dv.width = 32u; + dv.height = 1u; + dv.stride = 32u * 4u; + dv.format = SGX_FMT_A8R8G8B8; + dv.nlevels = 1; + dok = sgx_sampler_words(&dv, &dst, &dw0, + &dw1) == + SGX_SAMPLER_OK; + nu = ctx->range[k].fs ? + xpsb_attrib_issues( + &ctx->range[k].fs->attribs, + &ni, iof) : 0; + for (u = 0; u < nu && u < XPSB_MAX_TEX; + u++) { + unsigned ds = (unsigned)iof[u]; + + d[xpsb_texaddr_dw_at(0, ds)] = + (uint32_t) + sgx_ctx_va[SGX_CTX_BO_TEX]; + if (!dok || + SGX_ENV("SGX_NO_DUMMY_DESC")) + continue; + d[xpsb_texctl_dw_at(0, ds)] = + dw0; + d[xpsb_texstate_dw_at(0, ds)] = + dw1; + } + } + for (t = 0; t < SGX_MAX_TEX_UNITS; t++) { + unsigned ds; + + if (!ctx->range[k].tex[t].bound) + continue; + ds = ctx_tex_ds_fs(ctx->range[k].fs, + t); + d[xpsb_texctl_dw_at(0, ds)] = + ctx->range[k].tex[t].w0; + d[xpsb_texstate_dw_at(0, ds)] = + ctx->range[k].tex[t].w1; + d[xpsb_texaddr_dw_at(0, ds)] = + ctx->range[k].tex[t].va; + /* SGX_TEX1_ADDR points the second unit + * at an address of the caller's + * choosing. A fetch that never runs + * cannot fault, so an unmapped address + * there tells a second DOUTT that + * executes and loses its texel from one + * that never executes at all. */ + if (t == 1) { + const char *e = + SGX_ENVS("SGX_TEX1_ADDR"); + + if (e && *e) + d[ds] = + (uint32_t)strtoul(e, + NULL, 0); + } + } + if (SGX_ENV("SGX_DUMP_PDS")) { + const uint32_t *fr = (const uint32_t *) + ((const char *)heap + + ctx_rec_pri_pds_off(ctx, k)); + unsigned z; + + /* The frame's own program beside the + * record's copy of it: on this path the + * record runs the copy, so anything + * that differs beyond the patched + * address is a divergence. */ + fprintf(sgx_log(), "sgx: pds frame :"); + for (z = 0; z < 32; z++) + fprintf(sgx_log(), " %s%08x", + z % 4 ? "" : "| ", + fr[z]); + fprintf(sgx_log(), "\n"); + } + if (SGX_ENV("SGX_DUMP_PDS")) { + unsigned q; + + fprintf(sgx_log(), "sgx: pds data k=%u:", k); + for (q = 0; q < 32; q++) + fprintf(sgx_log(), " %s%08x", + q % 4 ? "" : "| ", + d[q]); + fprintf(sgx_log(), " (want tex va %08x)\n", + ctx->range[k].tex[0].va); + } + /* Whatever the record ends up naming, a zero + * here is the address the texture cache reads + * when the core stops. Say which record it + * was rather than leaving the fault report to + * name only the address. */ + if (SGX_ENV("SGX_CHECK_TEX") && + !d[xpsb_texaddr_dw_at(0, + ctx_tex_ds_fs(ctx->range[k].fs, 0))]) + fprintf(sgx_log(), + "sgx: record %u: zero texture address (has_tex %d, tex_va %08x, nviews %u)\n", + k, ctx->range[k].tex[0].bound, + ctx->range[k].tex[0].va, + ctx->nviews); + /* The second unit's slots sit two dwords past + * the first's, which is measured rather than + * assumed: with both units bound the frame's + * PDS data reads + * ds0 | .. .. 03fe0000 6c09f04f | + * | 43fe0000 6009f04f .. .. | + * ds1 | .. .. 82400000 .. | + * | 83a00000 .. .. .. | + * so unit 0 is {2,3,10} and unit 1 {4,5,12}, + * the two addresses being the units' own. + * Patched only when the frame does not already + * name it, which keeps the common path exactly + * as it was. */ + /* Whenever the record has one, not only when it + * differs from the frame's: the frame is no + * longer where this program's texture state + * comes from. Each record builds its own + * primary program now, out of heap state that + * is zero for unit 1 - so skipping the patch + * left the unit sampling an all-zero + * descriptor, which reads back black. Unit 0 + * is patched unconditionally, which is why it + * was the one that worked. */ + if (SGX_ENV("SGX_DEBUG") && + ctx->range[k].fs && + ctx->range[k].fs->attribs.nset > 2) + fprintf(sgx_log(), "sgx: record %u: 3-set, " + "has_tex %d has_tex1 %d t1 %u " + "va1 0x%x\n", k, + ctx->range[k].tex[0].bound, + ctx->range[k].tex[1].bound, + ctx_tex_ds_fs(ctx->range[k].fs, + 1), + ctx->range[k].tex[1].va); + { + uint32_t uo = usse_next; + uint32_t uw = 0, ub = 0; + unsigned nt = 0, tf; + int r2 = ctx_put_program(ctx, k, uo, + &nt, &ub); + + if (r2) + return r2; + usse_next = uo + ub; + r2 = ctx_use_word(ctx, uo, &uw); + if (r2) + return r2; + d[0] = uw; + /* The count spans two dwords, the low + * five bits here and the rest in + * ds0[8]; masking it to this one + * wrapped a program needing 32 to + * none at all. */ + if (!sgx_temp_ok(nt)) + return -ENOSPC; + tf = sgx_temp_field(nt); + d[1] = (d[1] & ~0xf8000000u) | + (tf << 27); + d[8] = (d[8] & ~0x3fu) | + sgx_temp_field_hi(nt); + } + + /* The copy is what this record's texture unit + * reads, so a zero address here is the render + * reading address zero - the fault the frame's + * own check cannot see, because it looks at + * the frame's data and bounds itself by the + * views the caller bound. */ + if (SGX_ENV("SGX_DEBUG")) { + unsigned uq; + + for (uq = 0; uq < SGX_MAX_TEX_UNITS; + uq++) { + unsigned tq = ctx_tex_ds_fs( + ctx->range[k].fs, uq); + + if (d[xpsb_texaddr_dw_at(0, tq)]) + continue; + fprintf(sgx_log(), "sgx: record " + "%u unit %u: zero " + "texture address " + "(has_tex %d/%d, ds %u, " + "preiter %d)\n", k, uq, + ctx->range[k].tex[0].bound, + ctx->range[k].tex[1].bound, + tq, ctx->range[k].fs ? + (int)ctx->range[k].fs->tex_preiterated + : -1); + } + } + if (SGX_ENV("SGX_DUMP_PDS")) { + unsigned q0 = ctx_tex_ds_fs( + ctx->range[k].fs, 0); + unsigned q1 = ctx_tex_ds_fs( + ctx->range[k].fs, 1); + + fprintf(sgx_log(), "sgx: pds after k=%u: " + "u0 ctl %08x fmt %08x addr " + "%08x | u1 ctl %08x fmt %08x " + "addr %08x (has_tex %d/%d)\n", + k, + d[xpsb_texctl_dw_at(0, q0)], + d[xpsb_texstate_dw_at(0, q0)], + d[xpsb_texaddr_dw_at(0, q0)], + d[xpsb_texctl_dw_at(0, q1)], + d[xpsb_texstate_dw_at(0, q1)], + d[xpsb_texaddr_dw_at(0, q1)], + ctx->range[k].tex[0].bound, + ctx->range[k].tex[1].bound); + } + w = win + xpsb_heap_pds_dw(win, XPSB_PDS_W_PRI); + *w = (*w & 0xff000000u) | + (uint32_t)(((va + pds) >> 4) & 0x00ffffffu); + /* And its size, which the frame's program set + * and this record's own may not share: an + * issue list of a different length runs short + * or runs on. */ + if (rec_pri_dwords) + *w = (*w & 0x00ffffffu) | + ((rec_pri_dwords & 0xffu) << 24); + } + + /* And its own uniforms. The secondary program names + * the block it reads from in its first data dword, so + * a record that renders with different uniforms needs + * a copy of both. Sharing one block made every record + * of a frame render with the values the last draw + * happened to leave. */ + /* Only for a record that has fragment uniforms at all. + * A program with none - gears computes its colour in + * the vertex stage and passes it as a varying - must + * keep the frame's own secondary program: repointing + * it at a copy naming an empty block cost two of the + * three gears. */ + /* A literal pool counts as much as a uniform does. The + * pool shares the block the secondary program loads, + * so a record left on the frame's program never gets + * its constants - it reads whatever the bank last + * held. glmark2's ideas logo is exactly that: its + * colour is the literal vec4(0.5,0.4,0.7,1) in + * ideas-logo.frag and nothing else, its uniform is + * unused and optimised away, and it rendered white. + * Cutting the shader down to that one constant showed + * it plainly - softpipe drew 128,102,178 and this + * driver drew 255,255,255 - while the same shader + * offscreen, in a frame of its own, was right. */ + { + const struct sgx_shader *rfs0 = ctx->range[k].fs ? + ctx->range[k].fs : + ctx->fs; + unsigned uni_dw = (ctx->range[k].fs_uni < + ctx->nfs_uni_set && ctx->fs_uni_set) ? + ctx->fs_uni_set_dwords[ctx->range[k].fs_uni] : 0u; + + if ((uni_dw || (rfs0 && rfs0->pool_dwords)) && + !SGX_ENV("SGX_NO_PER_RECORD_UNI")) { + unsigned set = ctx->range[k].fs_uni; + /* The record's own program. The frame's is a + * different one with a different uniform count + * and its own sampling mode, and a frame of + * ninety draws holds many. */ + const struct sgx_shader *rfs = rfs0; + /* Taken in order, not by record number: most + * records have no uniforms and reserving a + * kilobyte for each of them is what kept + * SGX_MAX_DRAWS at 4096. */ + unsigned ci = ctx->nconst_copy; + uint32_t cb = SGX_FS_CONST_COPY_OFF + + ci * SGX_FS_CONST_COPY_STEP; + uint32_t sp = SGX_SEC_PDS_COPY_OFF + + ci * SGX_SEC_PDS_COPY_STEP; + + if (ci >= SGX_MAX_CONST_COPIES) { + sgx_ctx_dbg("record %u wants the " + "%uth constant copy and " + "the frame holds %u\n", + k, ci + 1u, + SGX_MAX_CONST_COPIES); + return -ENOSPC; + } + ctx->nconst_copy++; + uint32_t *d, *ws; + + if (cb + SGX_FS_CONST_COPY_STEP > + ctx->bo[SGX_CTX_BO_HEAP].size || + sp + SGX_SEC_PDS_COPY_STEP > + ctx->bo[SGX_CTX_BO_HEAP].size) + return -ENOSPC; + /* The whole block first, then this record's + * uniforms over the front of it. Copying only + * the uniforms left the rest of the block - + * the sampler slots and the literal pool - + * zero, so a program that had both a uniform + * and a constant read its constants as zero + * and drew black. */ + memcpy((char *)heap + cb, scon, + SGX_FS_CONST_COPY_STEP); + if (uni_dw) { + unsigned nu = uni_dw; + + /* The same bound ctx_refresh_fs_constants() + * applies to the frame's block: a program + * that samples for itself keeps its + * sampler descriptors right behind its + * uniforms, and for one declaring none + * they start at dword zero - which this + * copy wrote the caller's constants over, + * so the sample read constant data. */ + /* Bounded by the record's own program, + * the same two bounds the frame's + * block gets: the pool sits behind the + * uniforms and the sampler slots + * behind those, and this copy is read + * with that program's offsets. */ + if (nu > rfs->pool_base) + nu = rfs->pool_base; + if (fs_samples(rfs)) { + unsigned lim = fs_smp_base(rfs, + ctx_smp_units(ctx, rfs)); + + if (nu > lim) + nu = lim; + } + if (nu) + ctx_pack_uniforms(rfs, + ctx->fs_uni_set[set], + ctx->fs_uni_set_dwords[set], + (uint32_t *)((char *) + heap + cb), nu); + } + /* And its own literal pool, where its own + * program put it. The block above is a copy of + * the frame's, which ctx_upload_pools() laid + * out for the frame's program - that program's + * pool at that program's pool_base. A record + * whose program chose a different base read + * its constants from wherever the frame's + * layout happens to have something, and one + * frame carries programs with bases 3, 7 and + * 16. Laid out by the record's own program, + * which is the same rule the frame's block + * follows rather than a second one. + * + * Skipped, not refused, when the pool would + * run past the block: failing here drops the + * whole frame, and a frame that renders with + * one record's constants stale beats no frame + * at all. Said once so it is not invisible. */ + if (rfs->pool_dwords) { + unsigned pend = rfs->pool_base + + rfs->pool_dwords; + + if (pend * 4u <= SGX_FS_CONST_COPY_STEP) + memcpy((uint32_t *)((char *) + heap + cb) + + rfs->pool_base, + rfs->pool, + rfs->pool_dwords * + sizeof(uint32_t)); + else if (nfit_said++ < 16u) + fprintf(sgx_log(), "sgx: record " + "%u: its literal pool " + "ends at dword %u, " + "past the %u its block " + "holds - it reads the " + "frame's constants\n", + k, pend, + (unsigned)(SGX_FS_CONST_COPY_STEP / 4u)); + } + /* The copy carries the frame's layout, so its + * sampler slots sit behind the frame program's + * uniforms. A record running a different + * program reads them behind its own, where + * that layout has constants - the texture + * address came out as the float 1.0 and the + * sample faulted on it. Written after the + * uniforms, which may cover the same dwords. */ + if (fs_samples(rfs)) + ctx_write_sampler_slots(ctx, rfs, + &ctx->range[k], + (uint32_t *)((char *)heap + cb), + cb); + d = (uint32_t *)((char *)heap + sp); + memcpy(d, ssec, SGX_SEC_PDS_COPY_STEP); + /* The relocation writes this dword as a plain + * heap address, so the copy is built the same + * way rather than shifted. */ + d[0] = (uint32_t)(va + cb); + /* The size field comes from what ctx_sec_pds() + * laid out, not from the block being copied: + * these copies are made before the frame's + * relocations run, so the descriptor still + * holds the captured template - which names a + * secondary program of size zero. Preserving + * that gave every copied record a zero-length + * secondary program, so its uniforms were + * never fetched and two of the three gears + * stopped rendering. */ + ws = win + xpsb_heap_pds_dw(win, XPSB_PDS_W_SEC); + *ws = ((ctx->sec_pds_dwords & 0xffu) << 24) | + (uint32_t)(((va + sp) >> 4) & 0x00ffffffu); + if (SGX_ENV("SGX_DEBUG")) { + const uint32_t *rb = + (const uint32_t *) + ((const char *)heap + cb); + + /* Read the copy back rather than the + * array it came from: that tells a + * copy holding the wrong values from + * one the hardware never reads. */ + fprintf(sgx_log(), "sgx: record %u: uni " + "set %u, %u dword(s), first " + "%08x, sec pds at 0x%x size " + "%u, block at 0x%x holds " + "%08x %08x %08x %08x, " + "smp slot %08x %08x %08x, " + "frame slot %08x %08x %08x, " + "nuniform %u, preiter %u, " + "secpds[0] %08x\n", k, set, + uni_dw, + uni_dw ? + ctx->fs_uni_set[set][0] : 0u, + sp, + ctx->sec_pds_dwords, cb, + rb[0], rb[1], rb[2], rb[3], + rb[4u * rfs->nuniform + 0], + rb[4u * rfs->nuniform + 1], + rb[4u * rfs->nuniform + 2], + ((const uint32_t *)((const char *)heap + + SGX_FS_CONST_OFF))[4u * rfs->nuniform + 0], + ((const uint32_t *)((const char *)heap + + SGX_FS_CONST_OFF))[4u * rfs->nuniform + 1], + ((const uint32_t *)((const char *)heap + + SGX_FS_CONST_OFF))[4u * rfs->nuniform + 2], + rfs->nuniform, + rfs->tex_preiterated, + d[0]); + { + const uint32_t *fo = (const uint32_t *) + ((const char *)heap + + XPSB_SEC_PDS_OFF); + unsigned z; + + /* The frame's secondary program + * beside the record's copy: the + * copy is what loads this + * record's sa bank. */ + fprintf(sgx_log(), "sgx: sec frame:"); + for (z = 0; z < 8; z++) + fprintf(sgx_log(), " %08x", fo[z]); + fprintf(sgx_log(), "\nsgx: sec copy :"); + for (z = 0; z < 8; z++) + fprintf(sgx_log(), " %08x", d[z]); + fprintf(sgx_log(), "\n"); + } + } + } else if (SGX_ENV("SGX_DEBUG")) { + fprintf(sgx_log(), "sgx: record %u: no uniforms " + "of its own (set %u of %u, %u dword(s))" + "\n", k, ctx->range[k].fs_uni, + ctx->nfs_uni_set, + ctx->range[k].fs_uni < ctx->nfs_uni_set && + ctx->fs_uni_set_dwords ? + ctx->fs_uni_set_dwords[ctx->range[k].fs_uni] : 0); + } + } + if (SGX_ENV("SGX_DUMP_STATEBLK") && k < 8) { + const uint32_t *o = (const uint32_t *) + ((const char *)heap + + SGX_STATE_WINDOW_OFF); + const uint32_t *c = (const uint32_t *) + ((const char *)heap + off); + unsigned q; + + /* The copy against what it was copied from: + * the two paths differ only in which of these + * the record points at. */ + fprintf(sgx_log(), "sgx: state orig k=%u:", k); + for (q = 0; q < 24; q++) + fprintf(sgx_log(), " %08x", o[q]); + fprintf(sgx_log(), "\nsgx: state copy k=%u:", k); + for (q = 0; q < 24; q++) + fprintf(sgx_log(), " %08x", c[q]); + fprintf(sgx_log(), "\n"); + } + r[0] = (tmpl[0] & ~0x0fffffffu) | + (uint32_t)(((va + off + SGX_STATE_IN_WINDOW) >> 4) & + 0x0fffffffu); + } + } + /* Rate limited: the condition holds for every frame of a client that + * meets it, and this is a diagnostic, not a log. */ + if (nbase_off && nfit_said++ < 16u) + fprintf(sgx_log(), "sgx: frame: %u of %u record(s) run a program " + "whose attribute bases differ from the frame's - reg " + "%u against %u\n", nbase_off, ctx->nrange, base_mine, + base_frame); + if ((nsa_short || nsplit_short) && nfit_said++ < 16u) + fprintf(sgx_log(), "sgx: frame: %u of %u record(s) reach past the " + "%u sa dword(s) the frame's secondary program loads, " + "%u past its register split - those draws render with " + "the bank's leftovers\n", nsa_short, ctx->nrange, + ctx->sec_sa_dwords, nsplit_short); + memcpy(v + (at + nrec) * XPSB_DRAW_STRIDE, term, sizeof term); + ctx->ndraw = at + nrec + 1; + /* The stream as the tiler will read it: the captured prefix, our + * records, then the terminator. A record the hardware is never told + * about looks exactly like one that rasterised nothing. */ + if (SGX_ENV("SGX_DUMP_RECORDS")) { + unsigned r, q; + + fprintf(sgx_log(), "sgx: records: prefix %u, at %u, nrange %u, " + "ndraw %u, cmd_off %u\n", ctx_ta_prefix(ctx), at, + ctx->nrange, ctx->ndraw, ctx->draw_cmd_off); + for (r = 0; r < ctx->ndraw; r++) { + fprintf(sgx_log(), "sgx: draw %u:", r); + for (q = 0; q < XPSB_DRAW_STRIDE; q++) + fprintf(sgx_log(), " %08x", + v[r * XPSB_DRAW_STRIDE + q]); + fprintf(sgx_log(), "\n"); + } + } + return 0; +} + +/* The vertex program and the uniforms it reads. The program goes where the + * retargeted USE records point; the uniforms into the hole the vertex PDS + * DMAs from. A caller whose vertex shader did not compile still gets a frame: + * the built-in transform runs in its place, which is what the frame was built + * with. */ +static int ctx_upload_vs(struct sgx_context *ctx) +{ + uint32_t *heap = ctx->bo[SGX_CTX_BO_HEAP].map; + uint32_t *use = ctx->bo[SGX_CTX_BO_USSE].map; + int ret; + + if (!heap || !use) + return -EINVAL; + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: upload vs: hwtcl %d, vs %p compiled %d, " + "fixed %d, nlimm %u, const %u\n", ctx->hwtcl, + (const void *)ctx->vs, ctx->vs ? ctx->vs->compiled : 0, + ctx->vs_fixed, ctx->vs ? ctx->vs->nlimm : 0, + ctx->vs_const_dwords); + /* SGX_NO_VS_UPLOAD leaves the frame's built-in transform in place, to + * tell a compiled program the frame runs from one it only stores. */ + static int novs = -1; + + if (novs < 0) + novs = SGX_ENVS("SGX_NO_VS_UPLOAD") != NULL; + if (novs) { + /* Neither program is written, so the USSE object keeps + * whatever the frame put there. */ + } else if (!ctx->vs_fixed && ctx->vs && ctx->vs->compiled && + ctx->vs != sgx_passthrough_vs()) { + ret = sgx_hwtcl_set_program(use, ctx->bo[SGX_CTX_BO_USSE].size, + ctx->vs->code, ctx->vs->ninsns); + if (ret) { + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: upload vs: the program " + "(%u insn) would not fit: %d\n", + ctx->vs->ninsns, ret); + return ret; + } + ret = sgx_hwtcl_set_temps(heap, ctx->vs->ntemps); + if (ret) { + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: upload vs: %u temp(s) " + "refused: %d\n", ctx->vs->ntemps, ret); + return ret; + } + /* One copy of the program per uniform set the frame asked + * for, each with those values compiled into it. The records + * are pointed at their own copy in ctx_emit_draw_records(). */ + if (ctx->vs->nlimm) { + unsigned k, nset = ctx->nuni_set ? ctx->nuni_set : 1; + + /* One copy of the program per uniform set the frame + * asked for, each with those values compiled in, and + * each record bound to its own through a copy of the + * vertex PDS program. */ + for (k = 0; k < nset; k++) { + unsigned off = sgx_hwtcl_prog_off(k); + const float *v = ctx->nuni_set ? + ctx->uni_set[k] : ctx->vs_const; + unsigned n = ctx->nuni_set ? + ctx->uni_set_dwords[k] : + ctx->vs_const_dwords; + + ret = sgx_hwtcl_set_program_at( + use, ctx->bo[SGX_CTX_BO_USSE].size, + off, ctx->vs->code, ctx->vs->ninsns); + if (ret) + return ret; + ret = sgx_hwtcl_patch_uniforms(use, off, + ctx->vs->limm, + ctx->vs->nlimm, + v, n); + if (ret) + return ret; + } + } + } else { + sgx_hwtcl_default_program(use); + sgx_hwtcl_set_temps(heap, 0); + } + if (SGX_ENV("SGX_DUMP_VPDS")) { + const uint32_t *d = heap + SGX_HWTCL_VPDS_DATA_OFF / 4; + unsigned q; + + /* The frame's own vertex PDS program: twelve data dwords then + * its code. A uniform DMA has to be added to this. */ + fprintf(sgx_log(), "sgx: vertex pds:"); + for (q = 0; q < 20; q++) + fprintf(sgx_log(), " %s%08x", q % 4 ? "" : "| ", d[q]); + fprintf(sgx_log(), "\n"); + } + if (SGX_ENV("SGX_DEBUG")) { + unsigned q; + + fprintf(sgx_log(), "sgx: vs uniforms: %u dwords, limm %u, %u " + "insn(s), %u out, %u temps, prog %s:", + ctx->vs_const_dwords, ctx->vs ? ctx->vs->nlimm : 0, + ctx->vs ? ctx->vs->ninsns : 0, + ctx->vs ? ctx->vs->nvtxout : 0, + ctx->vs ? ctx->vs->ntemps : 0, + (!ctx->vs_fixed && ctx->vs && ctx->vs->compiled && + ctx->vs != sgx_passthrough_vs()) ? "caller" : "builtin"); + for (q = 0; q < ctx->vs_const_dwords && q < 8; q++) + fprintf(sgx_log(), " %g", ctx->vs_const[q]); + fprintf(sgx_log(), "\n"); + } + /* A compiled vertex program with no uniforms is not an error: there is + * simply nothing to upload, and the MTE applies the viewport. This + * refused the whole frame instead, so every draw in it disappeared and + * the caller saw one -EINVAL from the flush with nothing to say which + * of a dozen steps produced it. A shader that only passes its + * attributes through is exactly this shape, and it is what three + * samplers on one coordinate set produce. */ + if (!ctx->vs_const_dwords) + return 0; + if (sgx_hwtcl_limm_uniforms()) + return 0; /* the values are in the code */ + if (sgx_hwtcl_inrec_uniforms()) + return 0; /* the values are in the record */ + return sgx_hwtcl_set_uniforms(heap, sgx_ctx_va[SGX_CTX_BO_HEAP], + ctx->vs_const, ctx->vs_const_dwords); +} + +/* A texture address of zero in the primary PDS is what the render reads at + * address zero, with the texture cache named as the requestor. Nothing that + * builds one should leave a zero behind, so say which slot did rather than + * letting the core stop and the fault report name only the address. */ +static void ctx_check_tex_addrs(struct sgx_context *ctx) +{ + const uint32_t *d; + unsigned u, ntex; + + if (!ctx->bo[SGX_CTX_BO_HEAP].map) + return; + d = (const uint32_t *)ctx->bo[SGX_CTX_BO_HEAP].map + + ctx_pri_pds_off(ctx) / 4; + /* Exactly what xpsb_heap_set_ntex() issued: a slot the frame does not + * issue may hold anything and that is not a fault. Bounding this by + * the coordinate sets instead reported an unused unit every frame. */ + ntex = ctx->hwtcl ? (unsigned)sgx_hwtcl_ntex() : ctx->nviews; + if (ntex > SGX_MAX_TEX_UNITS) + ntex = SGX_MAX_TEX_UNITS; + for (u = 0; u < ntex; u++) + if (!d[xpsb_texaddr_dw_at(0, ctx_tex_ds(ctx, u))]) + fprintf(sgx_log(), "sgx: frame pds: unit %u has a zero " + "texture address (nviews %u, bound 0x%x, " + "ctx tex2 va 0x%llx bound %d, tex_bo1 %p va 0x%llx, " + "pri_pds %u, nrange %u)\n", + u, ctx->nviews, ctx->view_bound, + (unsigned long long)ctx->bo[SGX_CTX_BO_TEX2].gpu_va, + ctx->bound[SGX_CTX_BO_TEX2], + (void *)ctx->tex_bo[1], + (unsigned long long)(ctx->tex_bo[1] ? + ctx->tex_bo[1]->gpu_va : 0), + ctx->pri_pds_dwords, ctx->nrange); +} + +/* How many submits a displayed frame costs. Every split - a viewport change, a + * different attribute layout, a switch of transform path - is a whole frame: + * its own setup, its own tiler pass and a render that loads back and stores + * every tile. So the submits-per-swap ratio is the first thing to know about + * this driver's speed, and nothing reported it. SGX_PERF=1 prints it. */ +static unsigned sgx_perf_submits; +/* Where a swap's wall time goes, in microseconds over the same 64 swaps as the + * split tally. Splits and frame rate turned out not to be proportional - + * removing a seventh of ioquake3's submits moved it five per cent - so the + * report has to say what the time is actually spent on rather than leave it to + * be inferred from a count. */ +static int sgx_perf_on(void); +static unsigned long sgx_perf_us_build, sgx_perf_us_wait, sgx_perf_us_flush; +static unsigned long sgx_perf_us_vtx, sgx_perf_us_rec, sgx_perf_us_draw; +static unsigned long sgx_perf_ndraw; +static unsigned long sgx_perf_us_wall; +static uint64_t sgx_perf_t_swap; + +static uint64_t sgx_perf_now(void) +{ + struct timespec ts; + + if (clock_gettime(CLOCK_MONOTONIC, &ts)) + return 0; + return (uint64_t)ts.tv_sec * 1000000ull + (uint64_t)ts.tv_nsec / 1000ull; +} + +void sgx_perf_add(int which, uint64_t t0) +{ + uint64_t d; + + if (!sgx_perf_on() || !t0) + return; + d = sgx_perf_now() - t0; + switch (which) { + case 0: sgx_perf_us_build += d; break; + case 1: sgx_perf_us_wait += d; break; + case 3: sgx_perf_us_vtx += d; break; + case 4: sgx_perf_us_rec += d; break; + case 5: sgx_perf_us_draw += d; sgx_perf_ndraw++; break; + default: sgx_perf_us_flush += d; break; + } +} + +uint64_t sgx_perf_mark(void) +{ + return sgx_perf_on() ? sgx_perf_now() : 0; +} +/* Grown with the number of distinct reasons rather than fixed: a reason that + * did not fit would be a submit the report cannot account for, which is the + * thing this is here to find. */ +static struct { const char *why; unsigned n; } *sgx_perf_reason; +static unsigned sgx_perf_nreason, sgx_perf_reason_max; + +static int sgx_perf_on(void) +{ + static int on = -1; + + if (on < 0) + on = SGX_ENV("SGX_PERF"); + return on; +} + +/* Which split ended the frame, so the submits-per-swap figure says what to + * fix rather than only how bad it is. Keyed on the literal, like the split + * tally, so a caller passes a string constant. */ +void sgx_perf_note(const char *why) +{ + unsigned i; + + if (!sgx_perf_on()) + return; + /* A client whose swap does not reach sgx_perf_swap() - the DRM + * platform's does not - never prints the tally, so SGX_PERF=2 + * says each reason as it happens instead. */ + if (atoi(SGX_ENVS("SGX_PERF")) > 1) + fprintf(sgx_log(), "sgx: split: %s\n", why); + for (i = 0; i < sgx_perf_nreason; i++) + if (sgx_perf_reason[i].why == why) { + sgx_perf_reason[i].n++; + return; + } + if (sgx_perf_nreason == sgx_perf_reason_max) { + unsigned want = sgx_perf_reason_max ? + sgx_perf_reason_max * 2u : 16u; + void *p = realloc(sgx_perf_reason, want * sizeof *sgx_perf_reason); + + if (!p) + return; + sgx_perf_reason = p; + sgx_perf_reason_max = want; + } + sgx_perf_reason[sgx_perf_nreason].why = why; + sgx_perf_reason[sgx_perf_nreason].n = 1; + sgx_perf_nreason++; +} + +void sgx_perf_swap(void) +{ + static unsigned swaps; + unsigned i; + + if (!sgx_perf_on()) + return; + if (++swaps % 64u) + return; + { + uint64_t now = sgx_perf_now(); + + if (sgx_perf_t_swap) + sgx_perf_us_wall = (unsigned long)(now - sgx_perf_t_swap); + sgx_perf_t_swap = now; + } + fprintf(sgx_log(), "sgx: perf: %u submit(s) over 64 swap(s), %.1f per " + "swap; wall %lu ms, draw %lu ms in %lu call(s) (vtx %lu), " + "build %lu ms (rec %lu), wait %lu ms, flush %lu ms;", + sgx_perf_submits, (double)sgx_perf_submits / 64.0, + sgx_perf_us_wall / 1000ul, sgx_perf_us_draw / 1000ul, + sgx_perf_ndraw, + sgx_perf_us_vtx / 1000ul, sgx_perf_us_build / 1000ul, + sgx_perf_us_rec / 1000ul, + sgx_perf_us_wait / 1000ul, sgx_perf_us_flush / 1000ul); + sgx_perf_us_build = sgx_perf_us_wait = sgx_perf_us_flush = 0; + sgx_perf_us_vtx = sgx_perf_us_rec = sgx_perf_us_draw = 0; + sgx_perf_ndraw = 0; + for (i = 0; i < sgx_perf_nreason; i++) { + fprintf(sgx_log(), " %u x %s;", sgx_perf_reason[i].n, + sgx_perf_reason[i].why); + sgx_perf_reason[i].n = 0; + } + fputc('\n', stderr); + sgx_perf_submits = 0; +} + +/* Put back what a damaged frame moved: the surface addresses and the extent + * the pixel back end may write. Both belong to the objects, not to the frame, + * so every way out of sgx_flush() has to run this. */ +static void ctx_damage_restore(struct sgx_context *ctx, uint64_t dmg_off, + uint32_t dmg_extent) +{ + if (dmg_off) { + ctx->bo[SGX_CTX_BO_TARGET].gpu_va -= dmg_off; + ctx->bo[SGX_CTX_BO_DEPTH].gpu_va -= dmg_off; + } + if (dmg_extent && ctx->bo[SGX_CTX_BO_HEAP].map) + ((uint32_t *)ctx->bo[SGX_CTX_BO_HEAP].map)[XPSB_HEAP_EXTENT] = + dmg_extent; +} + +static int sgx_flush_inner(struct sgx_context *ctx); + +/* The whole of ending a frame: the records, the submit and whatever wait goes + * with it. Timed against the frame build so the report can say which of the + * two a split actually costs. */ +int sgx_flush(struct sgx_context *ctx) +{ + uint64_t t = sgx_perf_mark(); + int r = sgx_flush_inner(ctx); + + sgx_perf_add(2, t); + return r; +} + +static int sgx_flush_inner(struct sgx_context *ctx) +{ + uint32_t *cmd; + unsigned nta, nta_ras, nras, noom; + unsigned sub_w, sub_h; + uint64_t dmg_off = 0, tr; + uint32_t dmg_extent = 0; + struct sgx_bo *bos[SGX_SUBMIT_MAX_BO]; + const struct sgx_shader *frame_fs; + const char *step = "building the frame"; + unsigned nh = 0, i; + int fence = -1; + int z_load = 0; + int ret; + + if (ctx && SGX_ENV("SGX_CHECK_TEX")) + ctx_check_tex_addrs(ctx); + if (ctx && ctx->frame_invalid) { + /* A draw was refused after some of this frame had already + * been built, so the records no longer describe the geometry. + * Drop it rather than hand the tiler a frame that disagrees + * with itself. */ + if (SGX_ENV("SGX_DEBUG") || SGX_ENV("SGX_HUD_TRACE")) + fprintf(sgx_log(), "sgx: flush: the frame was marked " + "invalid, dropping %u draw(s), %u range(s)\n", + ctx->draws, ctx->nrange); + sgx_context_drop_frame(ctx); + return -ENOSPC; + } + if (!ctx) + return -EINVAL; + /* A pending clear is work even with no draws: the frame's first record + * is the clear itself, so the stream is well formed without a user + * draw behind it. */ + if (!ctx->draws && !ctx->clear_pending) + return 0; /* nothing to do, and not an error */ + if (!ctx->cookie_valid) + return -EINVAL; + + /* The context holds the render target twice: by pointer, which the + * winsys keeps current through every bind, and by value, which is + * what the frame's relocations read. They must say the same address, + * and when they do not the frame is built against one and fired at + * the other - which the part reports as a fault inside the target's + * slot with nothing bound there, and which says nothing about the + * copy that went stale. + * + * The pointer is the one to believe. Corrected rather than refused, + * because a frame at the right address is worth more than a frame + * dropped - but said either way, because a copy that goes stale is a + * defect somewhere above this and the message is the only thread + * back to it. */ + if (ctx->target_bo && + ctx->bo[SGX_CTX_BO_TARGET].gpu_va != ctx->target_bo->gpu_va) { + fprintf(sgx_log(), + "sgx: the frame's target address 0x%llx is not where " + "the buffer is (0x%llx); using the buffer's\n", + (unsigned long long)ctx->bo[SGX_CTX_BO_TARGET].gpu_va, + (unsigned long long)ctx->target_bo->gpu_va); + ctx->bo[SGX_CTX_BO_TARGET].gpu_va = ctx->target_bo->gpu_va; + } + + /* The frame before this one may still be on the core, and everything + * below overwrites what it reads: the state windows and the PDS + * programs in the heap, the fragment code in the USSE buffer, the + * shader constants. This is the wait the submit no longer does, put + * where it costs the least - the application's draw calls for this + * frame have already run against the GPU rather than after it. */ + if (sgx_wait_idle(ctx->ws) && SGX_ENVS("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: flush: the frame before this one did not " + "finish; building this one anyway\n"); + + ctx_update_state(ctx); + + /* The programs' constants have to be in the sa bank before the code + * runs. Uploading them at flush rather than at bind is what makes two + * draws with the same shaders cost one upload. */ + ret = ctx_upload_pools(ctx); + if (ret) { + step = "uploading the literal pools"; + goto drop; + } + ret = ctx_emit_state(ctx); + if (ret) { + step = "emitting the state block"; + goto drop; + } + /* One program for the whole frame build. + * + * The iterator and the texture issues were derived from the program + * last bound while the code in the slot, and the secondary program + * that feeds it, were the first record's - so a frame carrying two + * fragment programs configured the two from different shaders and the + * fragment task waited on an issue the PDS never made. The frame + * serves record 0; the records behind it carry their own. */ + frame_fs = sgx_frame_fs(ctx); + ret = ctx_emit_samplers(ctx, frame_fs); + if (ret) { + step = "emitting the samplers"; + goto drop; + } + /* The vertex program emits the record, so its emit count is part of + * the record's width. The DMA, the byte stride and the group-10 field + * were all reprogrammed from the live width and this was not, so the + * captured program went on emitting eleven dwords into a record the + * hardware fetched at another stride. */ + if (ctx->bo[SGX_CTX_BO_USSE].map && ctx->vtx_floats) + xpsb_usse_set_vtx_dwords(ctx->bo[SGX_CTX_BO_USSE].map, + ctx->vtx_floats); + ret = ctx_upload_fs(ctx); + if (ret) { + step = "uploading the fragment program"; + goto drop; + } + + /* The transform's uniforms. They sit in a hole in the heap template + * the vertex PDS program DMAs into the attribute bank, so they are + * written per frame like any other frame content. */ + if (ctx->hwtcl) { + ret = ctx_upload_vs(ctx); + if (ret) { + step = "uploading the vertex program"; + goto drop; + } + } + + /* All four streams go into one command buffer at the offsets the + * captured frame used, because relocations address dwords within it. + * It stays on the host: the submit copies the two the kernel runs. */ + if (!ctx->cmd) { + ctx->cmd = calloc(1, SGX_CMD_BYTES); + step = "allocating the command buffer"; + if (!ctx->cmd) + { ret = -ENOMEM; goto drop; } + } + cmd = ctx->cmd; + sub_w = ctx->fb.width; + sub_h = ctx->fb.height; + nta = xpsb_gen_ta_stream(cmd + XPSB_TA_TA_OFF / 4); + nta_ras = xpsb_gen_ta_raster_stream(cmd + XPSB_TA_CMD_OFF / 4, + (int)ctx->fb.width, + (int)ctx->fb.height); + /* Multisampling, into both halves: the MTE has to iterate at the + * positions the ISP tested, so the two MULTISAMPLECTL registers carry + * the same word. Refused rather than ignored - a frame fired at a + * sample count only one half was built for renders at the wrong + * positions, which is a picture nobody can read as a fault. */ + { + unsigned smp = ctx->fb.samples ? ctx->fb.samples : 1u; + + if (xpsb_ta_set_msaa(cmd + XPSB_TA_TA_OFF / 4, nta, smp) || + xpsb_ta_raster_set_msaa(cmd + XPSB_TA_CMD_OFF / 4, nta_ras, + (int)ctx->fb.width, + (int)ctx->fb.height, smp)) { + /* Unconditionally: this drops the whole frame, and + * "returned -22" on its own does not say which + * geometry it refused. */ + fprintf(sgx_log(), "sgx: msaa: %ux%u at %u sample(s) " + "refused\n", ctx->fb.width, ctx->fb.height, + smp); + step = "programming the sample positions"; + ret = -EINVAL; + goto drop; + } + } + xpsb_ta_raster_set_bg(cmd + XPSB_TA_CMD_OFF / 4, nta_ras, + ctx->clear_depth, ctx->clear_stencil); + /* Goes with the background object the RC_BG relocation names: a frame + * that draws over an existing picture loads each tile back instead of + * clearing it. + * + * The object's extent decides nothing: measured on the part, a + * triangle covering the surface four times over paints exactly what + * the captured one does, and collapsing it to a point changes the + * picture completely. Its presence is what matters, not its shape. */ + xpsb_ta_raster_set_bg_load(cmd + XPSB_TA_CMD_OFF / 4, nta_ras, + !ctx->clear_on); + /* And the same for depth, or a pass that continues a frame starts every + * tile at the background depth and occludes nothing its predecessor + * drew - which is ioquake3's dynamic lights shining through walls, + * because their pass tests GL_EQUAL against depth a split left behind. + * Only over depth that is whole: a chain that never cleared has stored + * nothing to load. + * + * On over depth that is whole; SGX_NO_ZLOAD turns it off for comparison. + * xpsb_ta_raster_set_z_load emits the ZLSCTL load enable and clears the + * background object's mask plane, which together make the load survive - + * every case of driver/test/sgx_egl_depth.c renders. */ + z_load = ctx->depth_stored && !ctx->clear_on && !ctx->zclear_on && + !SGX_ENV("SGX_NO_ZLOAD"); + xpsb_ta_raster_set_z_load(cmd + XPSB_TA_CMD_OFF / 4, nta_ras, z_load); + sgx_ctx_dbg("zload %d (stored %d, clear %d, zclear %d)\n", z_load, + ctx->depth_stored, ctx->clear_on, ctx->zclear_on); + nras = xpsb_gen_raster_stream(cmd + XPSB_RAS_CMD_OFF / 4, + (int)ctx->fb.width, + (int)ctx->fb.height); + xpsb_ta_raster_set_bg(cmd + XPSB_RAS_CMD_OFF / 4, nras, + ctx->clear_depth, ctx->clear_stencil); + /* A frame that only touches part of an existing picture is built for + * that rectangle rather than for the whole surface: the scene, the + * extent and the tile range all come from it, and the target and depth + * addresses move to its origin. The geometry was moved into it as the + * record was written. + * + * This is what the ten milliseconds a frame were: the cost follows the + * scene, not the drawing inside it - the same fills cost 973 us of + * tiler and 10198 us of render in a screen-sized scene, and 60 us and + * 31 us in one sized to the damage. + * + * Not when the pass loads depth back: the ZLS addresses memory from the + * base registers by the pass's own extent and origin, so a rectangle + * that differs from the one the store used reads another layout - the + * full frame keeps depth and damage coherent instead. */ + if (ctx->dmg_on > 0 && ctx_damage_usable(ctx)) { + /* In sample tiles, which is what the render box counts when + * the frame is multisampled - the same rule + * xpsb_ta_raster_set_msaa() applies to the whole-surface box, + * and this rewrites that register, so it has to agree. */ + unsigned ax = xpsb_msaa_axis(ctx->fb.samples ? + ctx->fb.samples : 1u); + unsigned tw = ((ctx->dmg_w + 15u) / 16u) * ax; + unsigned th = ((ctx->dmg_h + 15u) / 16u) * ax; + + /* Only the tile range is in this stream. The pixel extent is + * SGX_S_TA_PIXEL_EXTENT, which the kernel programs from the + * scene cookie - so it follows sub_w/sub_h below, and the + * patcher that used to be called here scanned for an opcode + * the stream never carries and quietly did nothing. */ + /* Last tile, not tile count: 0x0410 is EUR_CR_ISP_RENDBOX2 and + * both ends of the box are inclusive. */ + xpsb_ta_raster_set_range(cmd + XPSB_TA_CMD_OFF / 4, nta_ras, + 0, 0, tw - 1u, th - 1u); + sub_w = ctx->dmg_w; + sub_h = ctx->dmg_h; + /* Both surfaces start at the rectangle's origin. Restored + * below, because the addresses belong to the objects. */ + dmg_off = ((uint64_t)ctx->dmg_y0 * ctx->stride + + ctx->dmg_x0) * 4u; + ctx->bo[SGX_CTX_BO_TARGET].gpu_va += dmg_off; + ctx->bo[SGX_CTX_BO_DEPTH].gpu_va += dmg_off; + /* And so does the extent the pixel back end may write. It is + * the only word that bounds the render at its far edge, and + * with the address moved the surface's own extent bounds + * nothing: every pixel from the rectangle's origin to the end + * of the surface is inside it. A rectangle whose width or + * height is not a whole number of tiles has an edge tile that + * reaches past it, and that is what wrote outside the + * rectangle. The stride is left alone - the rows are still the + * surface's rows apart. */ + if (ctx->bo[SGX_CTX_BO_HEAP].map) { + uint32_t *heap = ctx->bo[SGX_CTX_BO_HEAP].map; + + dmg_extent = xpsb_heap_extent(heap); + xpsb_heap_set_extent(heap, ctx->dmg_w, ctx->dmg_h); + } + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), + "sgx: damage: %ux%u at %u,%u of %ux%u\n", + ctx->dmg_w, ctx->dmg_h, ctx->dmg_x0, + ctx->dmg_y0, ctx->fb.width, ctx->fb.height); + } + noom = xpsb_gen_oom_stream(cmd + XPSB_TA_OOM_OFF / 4); + xpsb_ta_raster_set_bg(cmd + XPSB_TA_OOM_OFF / 4, noom, + ctx->clear_depth, ctx->clear_stencil); + if (!nta || !nta_ras || !nras) + { step = "generating the TA streams"; ret = -EINVAL; goto drop; } + + /* Every address in those streams is a placeholder until this runs. */ + { + struct xpsb_reloc tarel[XPSB_MAX_TA_RELOCS]; + struct xpsb_reloc rasrel[XPSB_NUM_RAS_RELOCS]; + struct xpsb_reloc_cfg cfg; + uint32_t *dstv[XPSB_FRAME_NBUF + 1]; + unsigned dstn[XPSB_FRAME_NBUF + 1]; + unsigned nrel, nrasrel; + + memset(&cfg, 0, sizeof cfg); + cfg.first_draw = 0u; + /* Without a clear the target has to survive, so the tiles the + * tiler did not bin load back instead of clearing. */ + cfg.bg_load = !ctx->clear_on; + if (SGX_ENV("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: frame: %ux%u bg_load %u " + "(clear_on %u), %u draw(s)\n", ctx->fb.width, + ctx->fb.height, cfg.bg_load, ctx->clear_on, + ctx->draws); + /* One unit unless the part runs the transform. + * + * The heap's own count follows nviews, but this one cannot yet: + * a second texture needs the wider vertex record - fourteen + * floats carrying a second coordinate set - and the software + * transform path emits eleven. Raising it alone makes the + * frame refuse to build. That record width is what a second + * texture unit is really waiting on. */ + /* The issue list says how many units there are and where each + * one's address slot sits: two units may share a coordinate + * set, and an iterated varying takes an issue of its own, so + * neither follows the record width the way this once assumed. */ + cfg.ntex = ctx->hwtcl ? 2u : 1u; + cfg.nissue = 0; + /* Whatever the builder chose, and unconditionally: the frame + * may be laid out from synthesised attributes, and a + * relocation left naming the captured slot would patch a + * program that is not the one running. Zero while it is where + * the capture had it, so that stream is unchanged. */ + if (ctx_pri_pds_off(ctx) != XPSB_PRI_PDS_OFF) + cfg.pri_pds_off = ctx_pri_pds_off(ctx); + /* The frame's program, the one whose code is in the slot - + * building the issue list from the program last bound named + * units the code never samples and left the frame's own + * texture slot unfilled, which the render then read at + * address zero. */ + /* A setless program's list too - the TAG issue attrib_issues() + * gives it - so the address slots follow what the primary + * program was actually built with. */ + /* Both transform paths, because xpsb_heap_set_attribs() lays + * the primary PDS out from this same issue list whichever one + * runs. Gating it on the software path left the hardware one + * relocating by unit index into a region addressed by issue, + * so the texture issue of a two-set frame kept its + * placeholder and the render faulted on 0x804ea000. */ + if (frame_fs) { + unsigned nu = xpsb_attrib_issues(&frame_fs->attribs, + &cfg.nissue, + cfg.tex_issue); + + if (nu) { + /* The generator binds at most the two units + * the capture sampled, so ntex stays clamped - + * refusing failed the whole frame the moment a + * program sampled a third unit. But every + * address slot still has to be relocated: + * a slot past the clamp kept the placeholder + * 0x804ea000, which is bound nowhere, and the + * render faulted on it with the cache as the + * requestor. ntexslot carries the real count + * so those slots get unit 0's mapped buffer. + * The first record reads the frame's own PDS + * data, so the per-record copies do not cover + * this. The hardware path binds the two units + * it declares whatever the program samples. */ + if (!ctx->hwtcl) + cfg.ntex = nu > XPSB_NTEX_MAX ? + XPSB_NTEX_MAX : nu; + cfg.ntexslot = nu; + /* Zero while the program is where the capture + * had it, so the relocation stream for a + * frame that fits is unchanged. */ + cfg.pri_pds_dwords = ctx->pri_pds_dwords; + } else { + cfg.nissue = 0; + } + } + if (SGX_ENV("SGX_DUMP_CMD")) { + unsigned k; + + fprintf(sgx_log(), + "sgx: ntex %u nissue %u pri_pds %u nset %u hwtcl %d fs_use_off %u\n", + cfg.ntex, cfg.nissue, ctx->pri_pds_dwords, + frame_fs ? frame_fs->attribs.nset : 0, + ctx->hwtcl, ctx->fs_use_off); + for (k = 0; k < cfg.nissue; k++) + fprintf(sgx_log(), "sgx: issue[%u] %u\n", + k, cfg.tex_issue[k]); + } + /* Where the fragment program is, so the relocation names it by + * the right base register and offset rather than by the + * captured 0x100. */ + cfg.frag_use_off = ctx->fs_use_off; + cfg.frag_use_size = ctx->fs_use_size; + /* The state block as ctx_emit_state() laid it out. */ + { + const uint32_t *heap = ctx->bo[SGX_CTX_BO_HEAP].map; + + unsigned n = xpsb_state_dwords(heap[xpsb_heap_state_base(heap)]); + + cfg.state_off = xpsb_heap_state_base(heap) * 4u; + cfg.pds_dw = xpsb_heap_pds_dw(heap, 0); + if (!n || n > XPSB_USSE_STATE_MAX) + { step = "sizing the state block"; + ret = -EINVAL; goto drop; } + cfg.state_use_off = xpsb_usse_state_copy_off(n); + cfg.state_use_size = xpsb_usse_state_copy_size(n); + } + cfg.sec_pds_dwords = ctx_sec_pds(ctx); + ctx->sec_pds_dwords = cfg.sec_pds_dwords; + if (cfg.sec_pds_dwords) { + cfg.sec_pds_off = XPSB_SEC_PDS_OFF; + cfg.sec_const_off = SGX_FS_CONST_OFF; + } + nrel = xpsb_gen_ta_relocs(tarel, &cfg); + nrasrel = xpsb_gen_raster_relocs(rasrel, &cfg); + if (!nrel || !nrasrel) + { step = "generating the relocations"; + ret = -EINVAL; goto drop; } + + /* The indices do not fit the fourteen dwords the heap holds, + * so the one relocation that produces their address is + * retargeted into the vertex buffer, as xpsb_frame.c's + * write_relocs() does for the DRM path. */ + for (i = 0; i < nrel; i++) + if (tarel[i].reloc_op == XPSB_RELOC_OP_OFFSET && + tarel[i].buffer == XPSB_BUF_HEAP && + tarel[i].pre_add == 0x3e4 && + tarel[i].dst_buffer == XPSB_BUF_VTX) { + tarel[i].buffer = XPSB_BUF_VTX; + tarel[i].pre_add = XPSB_IDX_OFF; + } + + /* The vertex task's entry point is three USE records against + * the frame's own pass-through program. A transforming one + * does not fit that 32-byte slot, so they are pointed at the + * program sgx_hwtcl_install() wrote instead. + * + * Frame-wide, out of the relocation table, and this is what + * stops a frame holding records of both kinds: every record's + * vertex task is aimed at the transforming program, so a + * record the draw module already transformed is transformed + * twice. The record's own vertex PDS copy can say + * SGX_VPDS_NO_PROG, but the entry point it runs comes from + * here. Letting a mixed frame stand renders the scene as a + * texture atlas; the settle path above is safe only because + * the whole frame agrees. */ + if (ctx->hwtcl) { + unsigned patched = 0; + + for (i = 0; i < nrel; i++) + if ((tarel[i].reloc_op == XPSB_RELOC_OP_USE_REG || + tarel[i].reloc_op == XPSB_RELOC_OP_USE_OFFSET) && + tarel[i].buffer == XPSB_BUF_USSE && + tarel[i].pre_add == XPSB_USSE_VTX_OFF) { + tarel[i].pre_add = SGX_HWTCL_USSE_OFF; + patched++; + } + if (patched != 3) + { step = "patching the texture relocations"; + ret = -EINVAL; goto drop; } + } + + /* The draw records moved, so every relocation that writes + * into them moves with them: its destination is a dword + * index into the vertex buffer and the prefix is in front. */ + if (ctx_ta_prefix(ctx)) + for (i = 0; i < nrel; i++) + if (tarel[i].dst_buffer == XPSB_BUF_VTX && + tarel[i].where * 4u < XPSB_VTX_OFF) + tarel[i].where += ctx_ta_prefix(ctx); + + /* The draw records are the relocations' destination, and + * ctx_emit_draw_records() reads its template and its + * terminator back out of them - so a frame of two or more + * overwrites the slot the terminator comes from, and the next + * frame took a draw record for its terminator, never ended the + * stream, and walked the vertex fetch into page zero. A short + * frame after a long one inherited the long one's trailing + * records the same way. Rebuild them before they are patched, + * twenty-eight dwords a frame, and both go away. */ + if (ctx->bo[SGX_CTX_BO_VTX].map) { + uint32_t *dv = (uint32_t *)ctx->bo[SGX_CTX_BO_VTX].map + + ctx_ta_prefix(ctx); + /* Zeroed: put_draw_record() writes four of the + * terminator's seven dwords, and the copy below took + * the other three from the stack - three + * nondeterministic dwords per frame into the record + * the tiler stops on. */ + uint32_t fresh[(XPSB_NUM_DRAW + 1) * + XPSB_DRAW_STRIDE] = { 0 }; + + /* The terminator slot alone. The records ahead of it + * are the frame's own clearing pair and the caller's + * relocated record, and resetting those took the clear + * from 56% coverage down to 12%. */ + if (xpsb_gen_draw_records_n(fresh, 0u, 1u)) + memcpy(dv + (XPSB_USER_DRAW + 1) * + XPSB_DRAW_STRIDE, + fresh + (XPSB_USER_DRAW + 1) * + XPSB_DRAW_STRIDE, + XPSB_DRAW_STRIDE * 4u); + } + memset(dstv, 0, sizeof dstv); + memset(dstn, 0, sizeof dstn); + dstv[XPSB_BUF_HEAP] = ctx->bo[SGX_CTX_BO_HEAP].map; + dstn[XPSB_BUF_HEAP] = + (unsigned)(ctx->bo[SGX_CTX_BO_HEAP].size / 4); + dstv[XPSB_BUF_VTX] = ctx->bo[SGX_CTX_BO_VTX].map; + dstn[XPSB_BUF_VTX] = + (unsigned)(ctx->bo[SGX_CTX_BO_VTX].size / 4); + dstv[XPSB_BUF_RASTGEOM] = ctx->bo[SGX_CTX_BO_RASTGEOM].map; + dstn[XPSB_BUF_RASTGEOM] = + (unsigned)(ctx->bo[SGX_CTX_BO_RASTGEOM].size / 4); + dstv[XPSB_BUF_CMD] = cmd; + dstn[XPSB_BUF_CMD] = SGX_CMD_BYTES / 4; + + ret = ctx_apply_relocs(ctx, tarel, nrel, ctx_ta_list, + XPSB_FRAME_NBUF + 1, dstv, dstn); + if (ret) + goto drop; + ret = ctx_apply_relocs(ctx, rasrel, nrasrel, ctx_ras_list, + XPSB_RAS_LIST_LEN, dstv, dstn); + if (ret) { + step = "applying the raster relocations"; + goto drop; + } + /* The background objects as they now stand in memory, after + * every relocation that could have rewritten them. A region + * the load-back does not reach is either the record's or it is + * not, and this is what says which without guessing. */ + if (SGX_ENV("SGX_BG_DUMP")) { + fprintf(sgx_log(), "sgx: frame %ux%u, clear %d, " + "bg_load %d\n", ctx->fb.width, ctx->fb.height, + ctx->clear_on, !ctx->clear_on); + xpsb_rastgeom_dump(sgx_log(), + ctx->bo[SGX_CTX_BO_RASTGEOM].map); + } + + /* Two sampled units make the primary PDS program long enough + * to run into the secondary null program that sits right + * behind it, so the null one moves out of the way. It has to + * happen here: the binding word is relocated, so writing it + * when the frame was built would be undone, and the HALT has + * to be in place before the binding points at it. */ + /* After the relocations, because the program copies this + * frame's DOUTU words and the first of them carries an + * address the kernel writes. */ + /* Inside the block: splice the pair in after the first two + * words of the user record, once the relocations have written + * their addresses, so the move carries the patched words with + * it. Everything past the splice slides up two dwords. */ + if (ctx->hwtcl && sgx_hwtcl_ta_record() && ctx_ta_inblock() && + ctx->bo[SGX_CTX_BO_VTX].map) { + uint32_t *dv = ctx->bo[SGX_CTX_BO_VTX].map; + unsigned at = XPSB_USER_DRAW * XPSB_DRAW_STRIDE + 2u; + unsigned end = (ctx->ndraw + 1u) * XPSB_DRAW_STRIDE; + + if ((end + 2u) * 4u <= XPSB_VTX_OFF) { + memmove(dv + at + 2u, dv + at, + (end - at) * 4u); + ret = sgx_hwtcl_input_const_state( + ctx->bo[SGX_CTX_BO_HEAP].map, + sgx_ctx_va[SGX_CTX_BO_HEAP], + ctx->bo[SGX_CTX_BO_USSE].map, + sgx_ctx_va[SGX_CTX_BO_USSE], + dv + at, ctx->vs_const, + ctx->vs_const_dwords); + if (ret) + goto drop; + if (ctx->draw_cmd_off >= at * 4u) + ctx->draw_cmd_off += 2u * 4u; + if (SGX_ENV("SGX_DUMP_RECORDS")) { + unsigned z; + + fprintf(sgx_log(), "sgx: spliced at %u:", + at); + for (z = 12; z < 24; z++) + fprintf(sgx_log(), " %08x", dv[z]); + fprintf(sgx_log(), "\n"); + } + } + } + if (ctx_ta_prefix(ctx)) { + ret = sgx_hwtcl_input_const_state( + ctx->bo[SGX_CTX_BO_HEAP].map, + sgx_ctx_va[SGX_CTX_BO_HEAP], + ctx->bo[SGX_CTX_BO_USSE].map, + sgx_ctx_va[SGX_CTX_BO_USSE], + ctx->bo[SGX_CTX_BO_VTX].map, + ctx->vs_const, ctx->vs_const_dwords); + if (ret) + goto drop; + if (SGX_ENV("SGX_DUMP_VPDS")) { + const uint32_t *ip = (const uint32_t *) + ctx->bo[SGX_CTX_BO_HEAP].map + + SGX_HWTCL_SEC_PDS_OFF / 4; + unsigned q; + + fprintf(sgx_log(), "sgx: input-state pds:"); + for (q = 0; q < 20; q++) + fprintf(sgx_log(), " %s%08x", + q % 4 ? "" : "| ", ip[q]); + fprintf(sgx_log(), "\n"); + } + } + + if (ctx->hwtcl) { + uint32_t *heap = ctx->bo[SGX_CTX_BO_HEAP].map; + + /* Bind the secondary program at heap+0x400, where the + * primary cannot reach it: at two coordinate sets the + * primary's code runs to 0x394 and would otherwise + * overwrite the null program the template keeps at + * 0x380. + * + * Only the binding. sgxtri writes a bare HALT here as + * well, because on bare metal nothing else is at + * 0x400 - but in this driver that is where + * ctx_sec_pds() has already built the real secondary + * program, whose DOUTD feeds the fragment task its + * constants. Stamping a HALT over its first word cost + * three quarters of every primitive's pixels. */ + heap[xpsb_heap_pds_dw(heap, XPSB_PDS_W_SEC)] = + (uint32_t)(((sgx_ctx_va[SGX_CTX_BO_HEAP] + + XPSB_SEC_PDS_OFF) >> 4) & + 0x00ffffffu); + } + } + + tr = sgx_perf_mark(); + ret = ctx_emit_draw_records(ctx); + sgx_perf_add(4, tr); + /* The primary PDS region as the hardware will read it, not as it was + * built: the two are not the same if anything patches it later. */ + if (SGX_ENV("SGX_PDS_WROTE")) { + const uint32_t *h = ctx->bo[SGX_CTX_BO_HEAP].map; + unsigned q; + + if (h) { + fprintf(sgx_log(), "sgx: pds at submit:"); + for (q = 0; q < 24; q++) + fprintf(sgx_log(), "%s %08x", + q % 8 ? "" : " |", + h[XPSB_PRI_PDS_OFF / 4 + q]); + fprintf(sgx_log(), "\n"); + /* The data-master word as submitted, so a forced + * pixel size can be seen to have survived. */ + fprintf(sgx_log(), "sgx: dms at submit: %08x " + "(pixelsize %u, usetask %u, pdstask %u)\n", + h[xpsb_heap_pds_dw(h, XPSB_PDS_W_CTL)], + h[xpsb_heap_pds_dw(h, XPSB_PDS_W_CTL)] & 0x7fu, + (h[xpsb_heap_pds_dw(h, XPSB_PDS_W_CTL)] >> 16) + & 3u, + (h[xpsb_heap_pds_dw(h, XPSB_PDS_W_CTL)] >> 25) + & 0x1fu); + } + } + if (!ret && SGX_ENV("SGX_DUMP_RECORDS")) { + const uint32_t *dv = ctx->bo[SGX_CTX_BO_VTX].map; + unsigned r, q; + + /* At submit, so the shortcut path shows the frame's own + * records and the per-record path shows the synthesised + * ones - the two are otherwise never comparable. */ + if (dv) { + fprintf(sgx_log(), "sgx: submit records: ndraw %u, " + "prefix %u, raw:", ctx->ndraw, + ctx_ta_prefix(ctx)); + for (q = 0; q < 8; q++) + fprintf(sgx_log(), " %08x", dv[q]); + fprintf(sgx_log(), "\n"); + for (r = 0; r < ctx->ndraw && r < 6; r++) { + fprintf(sgx_log(), "sgx: rec %u:", r); + for (q = 0; q < XPSB_DRAW_STRIDE; q++) + fprintf(sgx_log(), " %08x", + dv[r * XPSB_DRAW_STRIDE + q]); + fprintf(sgx_log(), "\n"); + } + } + } + if (!ret && SGX_ENV("SGX_DUMP_DRAW")) { + const uint32_t *v = ctx->bo[SGX_CTX_BO_VTX].map; + unsigned pre = ctx_ta_prefix(ctx), r, w; + + fprintf(sgx_log(), "sgx: tex0 va 0x%llx bound %d, target va 0x%llx\n", + (unsigned long long)ctx->bo[SGX_CTX_BO_TEX].gpu_va, + ctx->bound[SGX_CTX_BO_TEX], + (unsigned long long)ctx->bo[SGX_CTX_BO_TARGET].gpu_va); + fprintf(sgx_log(), "sgx: clear_on %d, prefix %u, ndraw %u, " + "cmd_off 0x%x, ranges %u\n", ctx->clear_on, pre, + ctx->ndraw, ctx->draw_cmd_off, ctx->nrange); + for (r = 0; v && r < ctx->ndraw && r < 5; r++) { + fprintf(sgx_log(), "sgx: rec %u:", r); + for (w = 0; w < XPSB_DRAW_STRIDE; w++) + fprintf(sgx_log(), " %08x", + v[pre + r * XPSB_DRAW_STRIDE + w]); + fputc('\n', stderr); + } + } + if (!ret && SGX_ENV("SGX_DUMP_STATE") && ctx->nrange > 1) { + const uint32_t *heap = ctx->bo[SGX_CTX_BO_HEAP].map; + const uint32_t *a = heap + SGX_STATE_WINDOW_OFF / 4; + const uint32_t *b = heap + (SGX_STATE_COPY_OFF + + SGX_STATE_WINDOW) / 4; + unsigned w, n = SGX_STATE_WINDOW / 4, diff = 0; + + /* The records point at the right addresses; this is whether + * what is at them is right. Record 1's window against the + * frame's own, dword for dword. */ + for (w = 0; w < n; w++) + if (a[w] != b[w]) { + if (diff < 12) + fprintf(sgx_log(), + "sgx: state[%03x] %08x -> %08x\n", + w * 4, a[w], b[w]); + diff++; + } + fprintf(sgx_log(), "sgx: state window: %u of %u dwords differ\n", + diff, n); + } + if (!ret && SGX_ENV("SGX_DUMP_RECORDS")) { + const uint32_t *v = ctx->bo[SGX_CTX_BO_VTX].map; + unsigned r, w; + + /* The parameter stream is the only thing left that differs + * between a frame that renders and one that stalls, so this + * prints it rather than reasoning about it. */ + for (r = 0; v && r < 6; r++) { + fprintf(sgx_log(), "sgx: rec %u:", r); + for (w = 0; w < XPSB_DRAW_STRIDE; w++) + fprintf(sgx_log(), " %08x", + v[r * XPSB_DRAW_STRIDE + w]); + fprintf(sgx_log(), "\n"); + } + } + if (ret) + return ret; + /* One frame's worth of what actually reaches the kernel, to diff a + * case that renders against one that does not. */ + /* One settled frame's worth of what actually reaches the kernel, to + * diff a case that renders against one that does not. */ + if (SGX_ENV("SGX_DUMP_CMD") && ctx->dumped_cmd++ == 3) { + unsigned k; + + for (k = 0; k + 1 < nta; k += 2) + fprintf(sgx_log(), "sgx: ta %04x %08x\n", + cmd[XPSB_TA_TA_OFF / 4 + k], + cmd[XPSB_TA_TA_OFF / 4 + k + 1]); + for (k = 0; k + 1 < nta_ras; k += 2) + fprintf(sgx_log(), "sgx: tras %04x %08x\n", + cmd[XPSB_TA_CMD_OFF / 4 + k], + cmd[XPSB_TA_CMD_OFF / 4 + k + 1]); + for (k = 0; k < SGX_CTX_BO_COUNT; k++) + fprintf(sgx_log(), "sgx: bo %u bound %d va %08x size %llu\n", + k, ctx->bound[k], (unsigned)ctx->bo[k].gpu_va, + (unsigned long long)ctx->bo[k].size); + for (k = 0; k < 16; k++) + fprintf(sgx_log(), "sgx: cookie[%u] %u\n", k, + ctx->cookie[k]); + } + for (i = 0; i < SGX_CTX_BO_COUNT; i++) + if (ctx->bound[i]) + bos[nh++] = &ctx->bo[i]; + /* The texture slots whether or not the context owns them. The frame's + * relocation names them by buffer index, and a buffer the submit does + * not carry cannot be resolved - the address is left at the captured + * placeholder or written as zero, and the render faults on it with the + * texture cache as the requestor. A caller's texture in the slot is + * not context-owned, so the loop above skips it, and a frame whose + * records happen to sample nothing never adds it either. */ + for (i = 0; i < SGX_CTX_BO_COUNT && nh < SGX_SUBMIT_MAX_BO; i++) { + unsigned q; + + if ((i != SGX_CTX_BO_TEX && i != SGX_CTX_BO_TEX2) || + !ctx->bo[i].gpu_va || !ctx->bo[i].handle) + continue; + for (q = 0; q < nh; q++) + if (bos[q] == &ctx->bo[i] || + bos[q]->handle == ctx->bo[i].handle) + break; + if (q == nh) + bos[nh++] = &ctx->bo[i]; + } + /* And every texture the frame samples that is not one of those. The + * render reads them after the whole scene is binned, so they have to + * stay mapped until then. */ + for (i = 0; i < ctx->nframe_tex && nh < SGX_SUBMIT_MAX_BO; i++) { + unsigned q; + + for (q = 0; q < nh; q++) + if (bos[q] == ctx->frame_tex[i]) + break; + if (q == nh) + bos[nh++] = ctx->frame_tex[i]; + } + + /* Everything the frame is built from - the heap, the vertex records, + * the shader code - is written through write-combining mappings, and + * a write-combining buffer is drained by a store fence on the CPU that + * filled it. The kernel fences before it fires, but by then the thread + * may be on another CPU and the buffer that holds the last of the + * frame is on the one it left. Fencing here, on the CPU that wrote it, + * is what makes the frame whole before the ioctl. */ + __builtin_ia32_sfence(); + /* The kernel's "raster" stream is the TA-raster one: it programs the + * ISP and the pixel back end for the render the tiler feeds. */ + sgx_perf_submits++; + /* What this frame names, so a later move of any of them knows to wait + * for it and everything else does not. */ + { + unsigned u; + + sgx_bo_mark_busy(ctx->ws, ctx->target_bo); + sgx_bo_mark_busy(ctx->ws, &ctx->bo[SGX_CTX_BO_TARGET]); + for (u = 0; u < SGX_MAX_TEX_UNITS; u++) + sgx_bo_mark_busy(ctx->ws, ctx->tex_bo[u]); + } + /* Only a frame that actually carries an armed object asks the kernel + * to read the counters back, so a frame with no query outstanding + * costs it nothing. */ + if (sgx_frame_has_vistest(ctx)) + sgx_vistest_arm(ctx->ws); + ret = sgx_submit_oom(ctx->ws, cmd + XPSB_TA_TA_OFF / 4, nta, + cmd + XPSB_TA_CMD_OFF / 4, nta_ras, + cmd + XPSB_TA_OOM_OFF / 4, noom, bos, nh, + sub_w, sub_h, &fence); + if (ret) { + /* The frame is gone either way, so it must not stay in the + * context. Leaving it there meant the next frame appended to + * a frame that was never submitted, and the one after that to + * both - so a single refusal grew without bound until + * something malformed reached the hardware. */ + ctx_damage_restore(ctx, dmg_off, dmg_extent); + ctx->dmg_on = 0; + /* The frame is gone and the next one is built from scratch, + * which is the whole of what a failed submit means here. + * Refusing every draw after it turned one bad frame into a + * client that stayed black for the rest of its life. */ + /* Reported before the frame is dropped, because dropping it is + * what zeroes the counts. */ + if (SGX_ENV("SGX_DEBUG") || SGX_ENV("SGX_HUD_TRACE")) + fprintf(sgx_log(), "sgx: submit failed with %d; %u draw(s)" + " in %u record(s) are lost and the next frame " + "is built from scratch\n", ret, ctx->draws, + ctx->nrange); + sgx_context_drop_frame(ctx); + return ret; + } + if (0) { + /* Anything that failed while the frame was being built leaves + * records that no longer describe the geometry, and the next + * flush used to append to them - the submit-failure path above + * already drops the frame for exactly that reason. */ +drop: + /* The rectangle's origin and extent belong to the frame, not + * to the objects. A step that fails after they were applied + * used to leave the bias on the addresses and the narrowed + * extent in the heap, so every later frame rendered into the + * dead rectangle. */ + ctx_damage_restore(ctx, dmg_off, dmg_extent); + /* Which step, not only that one did: a frame that fails to + * build reaches the caller as one number at the flush, and + * every step below returns the same -EINVAL. */ + { + static int said; + + if (!said) { + said = 1; + fprintf(sgx_log(), "sgx: the frame was not built: " + "%s returned %d - nothing it drew will " + "appear\n", step, ret); + } + } + sgx_context_drop_frame(ctx); + return ret; + } + + /* Whether the ISP wrote the depth surface at all: how many dwords of it + * are not the value the buffer was allocated with, after the render. */ + if (SGX_ENV("SGX_DEPTH_DUMP")) { + const uint32_t *d = ctx->bo[SGX_CTX_BO_DEPTH].map; + uint64_t n = ctx->bo[SGX_CTX_BO_DEPTH].size / 4u, q, nz = 0; + + for (q = 0; d && q < n; q++) + if (d[q]) + nz++; + fprintf(sgx_log(), "sgx: depth buffer %llu of %llu dwords set," + " [0..3] %08x %08x %08x %08x\n", + (unsigned long long)nz, (unsigned long long)n, + d ? d[0] : 0, d ? d[1] : 0, d ? d[2] : 0, + d ? d[3] : 0); + } + + if (SGX_ENV("SGX_DUMP_STATE")) { + const uint32_t *h = ctx->bo[SGX_CTX_BO_HEAP].map; + static const unsigned g[] = { XPSB_STATE_ISP_A, XPSB_STATE_ISP_B, + XPSB_STATE_ISP_C, XPSB_STATE_PDS, + XPSB_STATE_OUTSEL, + XPSB_STATE_CULL, + XPSB_STATE_TEXSIZE }; + unsigned q; + + /* After the submit, so the kernel's relocations are in - which + * is what the hardware ran, not what the driver wrote. */ + if (h) + fprintf(sgx_log(), "sgx: state %04x %08x (mask)\n", + xpsb_heap_state_base(h), + h[xpsb_heap_state_base(h)]); + for (q = 0; h && q < sizeof g / sizeof g[0]; q++) { + int at = xpsb_heap_state_off(h, g[q]); + + if (at >= 0) + fprintf(sgx_log(), "sgx: state %04x %08x\n", + at, h[at]); + } + /* SGX_DUMP_STATE=: dumps any heap window + * after the submit; the default is the vertex PDS data + * segment. */ + { + unsigned off = 0x3a0u, cnt = 24u; + + sscanf(SGX_ENVS("SGX_DUMP_STATE"), "%x:%u", &off, &cnt); + if (cnt > 256u) + cnt = 256u; + for (q = 0; h && q < cnt; q++) + fprintf(sgx_log(), "sgx: heap +0x%03x %08x\n", + off + q * 4u, h[off / 4u + q]); + } + } + ctx->last_fence = fence; + /* The streams as they were submitted, next to the picture the submit + * produced, so a frame that renders its draw and one that does not can + * be diffed rather than guessed at. Register/value pairs, which is how + * both streams are built. */ + { + const char *sp = SGX_ENVS("SGX_DUMP_STREAMS"); + + if (sp && *sp) { + static unsigned sn; + char nm[256]; + FILE *f; + + snprintf(nm, sizeof nm, "%s-%03u.txt", sp, sn++); + f = fopen(nm, "we"); + if (f) { + const uint32_t *ta = cmd + XPSB_TA_TA_OFF / 4; + const uint32_t *ra = cmd + XPSB_TA_CMD_OFF / 4; + unsigned q; + + fprintf(f, "box %.0f,%.0f..%.0f,%.0f " + "draws %u nidx %u nrange %u " + "vtx_floats %u hwtcl %d dmg %d " + "%ux%u+%u+%u sub %ux%u\n", + ctx->last_box[0], ctx->last_box[1], + ctx->last_box[2], ctx->last_box[3], + ctx->draws, ctx->nidx, ctx->nrange, + ctx->vtx_floats, ctx->hwtcl, + ctx->dmg_on, ctx->dmg_w, ctx->dmg_h, + ctx->dmg_x0, ctx->dmg_y0, sub_w, sub_h); + for (q = 0; q + 1 < nta; q += 2) + fprintf(f, "ta %04x %08x\n", + ta[q], ta[q + 1]); + for (q = 0; q + 1 < nta_ras; q += 2) + fprintf(f, "ras %04x %08x\n", + ra[q], ra[q + 1]); + /* What the part will actually fetch. Written + * here rather than logged per vertex: a line + * per vertex slows the client enough to change + * what the caller draws. */ + if (ctx->bo[SGX_CTX_BO_VTX].map && + ctx->vtx_floats) { + const char *vb = + ctx->bo[SGX_CTX_BO_VTX].map; + const float *v = (const float *) + (vb + XPSB_VTX_OFF); + const uint16_t *ix = (const uint16_t *) + (vb + XPSB_IDX_OFF); + unsigned n = ctx->vtx_float_cursor / + ctx->vtx_floats, e; + + for (q = 0; q < n; q++) { + fprintf(f, "vtx %u:", q); + for (e = 0; e < ctx->vtx_floats; + e++) + fprintf(f, " %g", + v[(size_t)q * + ctx->vtx_floats + + e]); + fputc('\n', f); + } + fprintf(f, "idx"); + for (q = 0; q < ctx->nidx; q++) + fprintf(f, " %u", ix[q]); + fputc('\n', f); + for (q = 0; q < ctx->nrange; q++) + fprintf(f, "range %u: first %u " + "count %u vtx_floats " + "%u objtype %u\n", q, + ctx->range[q].first, + ctx->range[q].count, + ctx->range[q].vtx_floats, + ctx->range[q].objtype); + } + if (ctx->bo[SGX_CTX_BO_HEAP].map) { + const uint32_t *h = + ctx->bo[SGX_CTX_BO_HEAP].map; + + for (q = 0; q < 0x1400; q++) + if (h[q]) + fprintf(f, "heap %03x " + "%08x\n", q, + h[q]); + } + fclose(f); + } + } + } + /* Every submit, including the ones a split makes - SGX_DUMP_RT is on + * the pipe flush and never sees those. */ + { + const char *dp = SGX_ENVS("SGX_DUMP_SUBMIT"); + const char *df = SGX_ENVS("SGX_DUMP_SUBMIT_FROM"); + const char *dt = SGX_ENVS("SGX_DUMP_SUBMIT_TO"); + static unsigned sq; + + /* A whole-target dump per submit fills the disk in one run, so + * a range keeps it to the submits under examination. */ + if (dp && *dp && (!df || !*df || sq >= (unsigned)atoi(df)) && + (!dt || !*dt || sq <= (unsigned)atoi(dt))) { + char nm[256]; + + snprintf(nm, sizeof nm, "%s-%03u-%ux%u+%u+%u.ppm", dp, + sq, ctx->dmg_on > 0 ? ctx->dmg_w : ctx->fb.width, + ctx->dmg_on > 0 ? ctx->dmg_h : ctx->fb.height, + ctx->dmg_on > 0 ? ctx->dmg_x0 : 0u, + ctx->dmg_on > 0 ? ctx->dmg_y0 : 0u); + ctx_damage_restore(ctx, dmg_off, dmg_extent); + dmg_off = 0; + dmg_extent = 0; + sgx_dump_target(ctx, nm); + } + if (dp && *dp) + sq++; + } + ctx_damage_restore(ctx, dmg_off, dmg_extent); + ctx->dmg_on = 0; + /* A clearing pass covered every region with its background object, so + * every region's depth was stored and the buffer is whole. A damage + * pass stored a different layout over part of it, so it is not. Any + * other pass leaves the answer as it was: what it touched it + * re-stored, what it left empty kept the depth already there. */ + if (ctx->clear_on || ctx->zclear_on) + ctx->depth_stored = 1; + else if (dmg_off) + ctx->depth_stored = 0; + sgx_context_drop_frame(ctx); + ctx->flushes++; + return 0; +} diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_context.h mesa-26.2.2/src/gallium/drivers/sgx/sgx_context.h --- mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_context.h 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/sgx_context.h 2026-09-08 10:57:36.679794220 +0200 @@ -0,0 +1,783 @@ +/* The driver-side context: what a Gallium pipe_context holds and does, in + * plain structs. + * + * This is the layer that has been missing between the reverse engineering and + * a Mesa driver. sgx_state.c translates one piece of state at a time; this + * decides what a draw and a flush actually do with the results - which objects + * a frame needs, in which address window, and what reaches the kernel. + * + * No Mesa tree is required. The fields mirror Gallium's names so wiring a real + * pipe_context to it is assignment rather than translation, and the whole thing + * is testable on the host today, which a real pipe_context would not be. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#ifndef _SGX_CONTEXT_H_ +#define _SGX_CONTEXT_H_ + +#include +#include + +#include "sgx_state.h" +#include "sgx_shader.h" +#include "sgx_sampler.h" +#include "sgx_winsys.h" +#include "sgx_hwtcl.h" +#include "xpsb_frame.h" + +/* The objects a frame needs, one per address window. The SGX reaches these + * through different requestors with different base registers, so which window + * an object lives in is a property of what it is, not an allocator choice. */ +enum sgx_ctx_bo { + SGX_CTX_BO_PARAM, /* the parameter heap the DPM fills */ + SGX_CTX_BO_HEAP, /* state, PDS programs and USSE code */ + SGX_CTX_BO_TARGET, /* the render target */ + SGX_CTX_BO_RASTGEOM, /* raster geometry */ + SGX_CTX_BO_USSE, /* shader code, its own requestor window */ + SGX_CTX_BO_VTX, /* draw records, vertices and indices */ + SGX_CTX_BO_DEPTH, /* depth and stencil */ + SGX_CTX_BO_TEX, /* texture data */ + SGX_CTX_BO_TEX2, /* the second unit, which hardware TCL's record needs */ + SGX_CTX_BO_COUNT +}; + +/* No colour buffer bound. Not zero: zero is SGX_FMT_A8, and testing the + * format for truth made an eight-bit target - the X server's glyph mask - + * indistinguishable from an unset one, so the frame programmed 8888 over it + * and wrote four bytes a pixel into a one-byte surface. */ +#define SGX_CBUF_FMT_NONE (~0u) + +struct sgx_framebuffer { + uint32_t width, height; + uint32_t cbuf_format; + int has_zsbuf; + /* Samples a pixel: 1, or 4 for the part's 2x2 mode. Zero is read as + * one, so a caller that does not know about multisampling gets what + * it always got. */ + uint32_t samples; +}; + +/* The DPM parameter heap, not the scene. sgx_ta_mem_info() holds back + * 0x600 pages whatever is asked for and refuses anything at or below 0x61f, so + * a heap only becomes useful at 0x640 pages. The three byte figures are the + * same generous starting points sgxtri uses; they are not derived from a + * display-list encoding this project has. */ +#define SGX_DPM_RESERVE_PAGES 0x600u +#define SGX_DPM_TAIL_PAGES 0x40u +#define SGX_DPM_MIN_WORK_PAGES (SGX_DPM_TAIL_PAGES + 1u) +#define SGX_DPM_BYTES_PER_MTILE 4096u +#define SGX_DPM_BYTES_PER_PRIM 256u +#define SGX_DPM_PRIM_MTILE_SPREAD 4u +/* The floor a context starts at, before the render target's own budget is + * worked out. */ +#define SGX_DPM_DEFAULT_MAX_PRIMS 4096u +/* The parameter heap the kernel maps, mirrored from SGX_PARAM_BYTES in + * psb_sgx_render.c. Userspace does not choose it - it is fixed so that the + * hardware's one heap load per boot serves every client - but the driver has + * to know how big it is to know how many primitives a frame may bin. Keep the + * two in step. */ +#define SGX_PARAM_HEAP_PAGES 0x4000u /* 64 MiB */ +/* The task's temporary count is five bits, so this is the largest it can + * describe - see sgx_temp_field(). */ +#define SGX_TEMP_FIELD_MAX 96u +/* Temporaries kept back from a program so a blend appended to it has somewhere + * to stage: sgx_blend_build() takes at most three, and the closing move is + * redirected into a fourth. */ +#define SGX_BLEND_TEMP_HEADROOM 8u +/* Registers in the secondary attribute bank. Measured: a shader whose only + * variable is how many constants it reads is steady at 32 and unreliable at + * 36 - see scratchpad note and the `manyconst` feature test. */ +/* Registers the USE divides between the tasks it runs at once. Derived, not + * measured: the captured split of 24 serves a task of seventeen registers and + * not one of thirty-nine, which puts the pool between 408 and 936; 512 is the + * power of two in that range. */ +#define SGX_USE_REG_POOL 512u + +#define SGX_SA_MAX 128u +/* Registers a sampler's state block takes in the sa bank. Must match + * SMP_SLOT in the backend (tools/usse-cc/backend/usse_cg.c). */ +#define SGX_SMP_SLOT SGX_SMP_SLOT_REGS +/* The heap is mapped four times the estimate the kernel hands the DPM, as + * sgxtri's dpm_param_cap() does: the tiler overruns the estimate, and without + * mapped memory behind it that overrun is an MMU fault with no out-of-memory + * event to explain it. The DPM packs page indices in 16 bits, which is what + * bounds the cap. */ +#define SGX_DPM_CAP_MULTIPLIER 4u +/* Every context maps at least this much parameter heap, whatever its render + * target needs. The DPM's heap load is a handshake the hardware takes once per + * boot, so the first client to submit fixes the budget for every client after + * it - and a heap sized from a 128x128 target then starved a 640x480 one. + * Sizing them all alike is what makes that single load serve everybody. */ +#define SGX_DPM_MIN_CAP_PAGES 0x4000u /* 64 MiB */ +#define SGX_DPM_MAX_PAGE_INDEX 0x10000u + +/* Room for all four streams at the offsets the captured frame used. */ +#define SGX_CMD_BYTES (XPSB_RAS_CMD_OFF + 4096u) + +/* What a draw needs from the caller's vertex state. The hardware's vertex + * record is a fixed 11 floats - position, colour, one coordinate set - so an + * element is described by where its floats are and how many there are. */ +#define SGX_MAX_VERTEX_ELEMENTS 8 +/* Draw records per frame. Each is a primitive block and the tiler allocates + * parameter memory per block, so these are groups of draws that share their + * state rather than one per draw. */ +/* Where the record array starts and how far it may grow. The ceiling is not + * the hardware's - that is the room the records take in the vertex buffer, + * checked when they are emitted - it is a bound on runaway growth. */ +#define SGX_DRAWS_INITIAL 64 +#define SGX_MAX_DRAWS 6144 +#define SGX_VTX_FLOATS 11 +/* With the vertex program on the USSE the record grows to fourteen: the + * program names pa14 upwards for its uniforms, so everything below has to be + * the record, and two sampled units are what produce that stride. */ +#define SGX_VTX_FLOATS_HWTCL 14 + +struct sgx_vertex_element { + unsigned src_offset; /* bytes into the vertex */ + unsigned vb_index; + unsigned ncomp; /* float components, 1 to 4 */ + unsigned stride; /* bytes between vertices */ + unsigned slot; /* first hardware float: 0 pos, 4 colour, 8 uv */ + /* The source's fourth float is 1/w, the draw module's post-viewport + * leftover. It goes into the record's fourth position float verbatim: + * that float is the RHW plane every iteration is corrected by, which + * is the vendor's own convention (Mesa vf_generic's viewport emit + * writes it the same way). 1.0 there iterates everything affine. */ + unsigned char rhw; + /* The source's components are half floats, two bytes each, and are + * widened on the way into the record. The record itself stays f32: + * the MTE reads a vertex as floats and its coordinate-set group names + * a component count and no format (EURASIA_MTE_TEXDIM_*, + * sgxdefs.h:2061-2065), so a narrow record is only reachable on the + * transform path, where the fetch lands in the vertex program's + * primary attributes instead. */ + unsigned char half; +}; + +/* IEEE binary16 to binary32, denormals and infinities included. The suite + * feeds values a wrong width or a wrong byte offset cannot round to, so this + * has to be exact rather than a fast approximation. */ +static inline float sgx_half_to_float(uint16_t h) +{ + union { uint32_t u; float f; } o; + uint32_t s = (uint32_t)(h & 0x8000u) << 16; + uint32_t e = ((uint32_t)h >> 10) & 0x1fu, m = (uint32_t)h & 0x3ffu; + + if (!e) { + int ex = -1; + + if (!m) { + o.u = s; + return o.f; + } + do { + ex++; + m <<= 1; + } while (!(m & 0x400u)); + o.u = s | ((uint32_t)(112 - ex) << 23) | ((m & 0x3ffu) << 13); + return o.f; + } + if (e == 31u) { + o.u = s | 0x7f800000u | (m << 13); + return o.f; + } + o.u = s | ((e + 112u) << 23) | (m << 13); + return o.f; +} + +/* Bytes one component of an element occupies. */ +static inline unsigned sgx_vertex_comp_size(const struct sgx_vertex_element *v) +{ + return v->half ? 2u : 4u; +} + +struct sgx_vertex_buffer { + const void *data; /* the mapped resource */ + unsigned offset, stride; + uint64_t size; +}; + +/* Where a fragment program's constants sit in the heap before the secondary + * PDS program moves them into the secondary attribute bank. Nothing else in + * the heap or the relocations names this block. */ +#define SGX_FS_CONST_OFF 0x600u + +/* How much of that block the driver will fill: the uniforms, the sampler + * state blocks between them and the literal pool, and the pool itself. */ +#define SGX_MAX_FS_CONST_DW 256u +/* The longest transfer one DOUTD control word describes: fifteen lines + * of sixteen dwords. */ +#define SGX_SEC_PDS_MAX_DW 240u +/* The square the context's own texture buffer is described as when a unit has + * no caller texture. It has to fit sgx_ctx_size[SGX_CTX_BO_TEX]. */ +#define SGX_CTX_TEX_DIM 64u +/* Distinct fragment-uniform sets one frame may carry. The uniforms live in a + * block the secondary PDS reads, and a record that renders with different ones + * needs its own block. The store grows as a frame asks for more: a fixed + * thirty-two refused the draw that went past it, and a refused draw drops the + * whole frame. A set cannot outnumber the records that name them, so that is + * the ceiling. */ +#define SGX_FS_MAX_UNI_SETS ((unsigned)SGX_MAX_DRAWS) +/* What the two uniform-set stores start at before a frame asks for more. */ +#define SGX_UNI_SETS_INITIAL 16u + +struct sgx_context { + struct sgx_winsys *ws; + struct sgx_bo bo[SGX_CTX_BO_COUNT]; + int bound[SGX_CTX_BO_COUNT]; + + struct sgx_framebuffer fb; + struct sgx_dsa_state dsa; + struct sgx_rasterizer_state rast; + + /* what the state translation produced, recomputed on bind rather than + * on draw - Gallium binds state objects far less often than it draws */ + struct sgx_isp_set isp_ff, isp_bf; + uint32_t raster_word; + unsigned colormask; /* the bound blend's PIPE_MASK_* set */ + /* The largest extent any descriptor has named at each unit's fixed + * address, which is what the fallback object there has to cover. */ + uint64_t tex_need[SGX_MAX_TEX_UNITS]; + float vp_scale[3], vp_translate[3]; + int vp_set; + int isp_dirty; + /* Which of the ISP's eight visibility counters the draws that follow + * increment, or -1 for none. One field per object, so one live query: + * a record cannot count into two registers at once. It goes into every + * record built while it is set, which is what makes a query that spans + * a frame split still count - each split's records carry it, and the + * kernel sums each render's contribution. */ + int vis_reg; + + /* the scene cookie, from the framebuffer size */ + uint32_t cookie[16]; + uint32_t scene_size, clear_start, clear_pages; + uint32_t max_prims; /* the largest scene a frame may bin */ + uint32_t stride; /* render target stride in pixels */ + unsigned ndraw, draw_cmd_off, nidx; + unsigned range_dropped; + /* Latched when a submit fails. This part does not recover from a + * malformed frame - the core stops and the machine goes with it - so + * once one frame has been refused the context stops offering more + * rather than resubmitting into it every frame. */ + + /* The scissor rectangle, already in the screen space the vertex + * records use - y downward, pixels. Applied by clipping geometry, + * because this part has no scissor register anyone has found. */ + /* The frame is built for this rectangle rather than the whole surface: + * the scene, the extent and the target base all follow it, and the + * geometry is moved into it as the record is written. */ + unsigned dmg_x0, dmg_y0, dmg_w, dmg_h; + float last_box[4]; /* the last draw's bounding box, for dumps */ + int dmg_on; + + int scissor_on; /* a rectangle has been set */ + int scissor_enable; /* and the rasterizer asks for it */ + float scissor_x0, scissor_y0, scissor_x1, scissor_y1; + /* Set when a draw this frame was refused. The frame no longer matches + * what the caller asked for, so it is not submitted. */ + int frame_invalid; + + /* One uniform set per distinct value the frame's draws asked for. + * A record names the set it was opened with, and each set gets its own + * copy of the vertex program with those values compiled into it. */ + float (*uni_set)[SGX_HWTCL_MAX_UNI_LIMM]; + unsigned *uni_set_dwords; + unsigned nuni_set, uni_set_cap; + /* One entry per group of draws that share their state: where its + * indices start and how many there are. */ + struct sgx_ctx_range { + unsigned first, count; + unsigned uni; /* which uniform set it was opened with */ + /* The record's own vertex stride. The vertex fetch's DMA + * control and byte stride live in the vertex PDS program, so + * a record that carries its own copy carries its own width + * and the frame no longer has to end when one differs. */ + unsigned vtx_floats; + /* Whether the part transformed this record, and whether the + * program it ran carries literal immediates - the frame's own + * answer is the last draw's, which is not this record's once a + * frame may hold both kinds. */ + unsigned char vpds_prog; + /* And whether the part transformed it at all, which decides + * the record's own MTE viewport: the draw module hands over + * window coordinates, so its records take the identity. */ + unsigned char hwtcl; + /* The ISP state the record was opened with: the front set and, + * under 2SIDED, the back one. */ + struct sgx_isp_set isp, isp_bf; + /* And its MTE control word - the cull face and shade model - + * with whether the group is emitted at all. */ + uint32_t cull; + int cull_on; + /* The shape the ISP rasterises from this record's vertices. + * The record's own, so one frame can hold a point block and + * a triangle block instead of ending between them. */ + unsigned objtype; + /* The texture this group samples, as the PDS data needs it: + * the two descriptor words and the address. A record after the + * first carries its own copy of that data, so these are what + * go into it. */ + /* One triple per texture unit: the frame's PDS data holds a + * pair of descriptor words and an address for each, and a + * record that samples several textures has to carry all of + * them. Once an array rather than a unit-0 triple and a + * unit-1 triple written out by hand - two units was never a + * hardware limit, and every place that spelled the second one + * out separately was a place the third had to be added to. + * + * The object that address belongs to goes with it, so the + * submit can name it: the kernel only keeps the buffers a + * submit lists mapped, and a texture the frame samples but + * does not list faults the render at its own address. */ + struct { + uint32_t w0, w1, va; + struct sgx_bo *obj; + int bound; + /* The planes behind va, for a program that samples + * them one state block at a time. */ + uint32_t nchunks, chunk_size; + } tex[SGX_MAX_TEX_UNITS]; + /* The program this group runs, and whether it is the blend + * operator's rather than the caller's. */ + const struct sgx_shader *fs; + enum sgx_blend_op blend; + struct sgx_blend_desc blend_desc; + uint32_t blend_const; + int blend_insn; + /* Whether this record blends at all. The named operator below + * is only the fallback for what the encoder cannot build, so + * gating on it alone left every factor pair outside that short + * list drawing unblended. */ + int blended; + /* A GL logic op rather than a blend equation. */ + unsigned logicop, logicop_func; + unsigned fs_uni; /* which fragment-uniform set */ + } *range; + unsigned nrange; + /* Records that have taken a per-record constant copy this frame. The + * copies used to be indexed by record number, so the heap reserved a + * kilobyte for every record whether it had constants or not - four + * megabytes of the five and a half the per-record regions have, for + * something most records never use. Handed out in order instead. */ + unsigned nconst_copy; + /* The record width this frame's vertices were written at. One frame + * names one width - the fields that carry it sit outside the state + * block each record copies - so a draw that needs another ends the + * frame rather than joining it at the wrong stride. */ + unsigned frame_vtx_floats; + /* Where the next record's vertices go, in floats from XPSB_VTX_OFF. + * Records carry their own stride now, so a vertex cannot be placed at + * its number times a frame-wide width: a narrower record would start + * inside the one before it. */ + size_t vtx_float_cursor; + /* The distinct texture objects this frame samples. The submit names + * them so the kernel keeps them mapped through the render, which + * happens long after the draws that asked for them. The kernel takes + * SGX_SUBMIT_MAX_HANDLES in one submit, so the frame ends when the + * budget is spent rather than when a texture changes. + * + * It was 48 against a kernel bound of 64, and ioquake3 spent one whole + * frame a swap on it - 64 splits in 64 swaps, measured. The bound is a + * sanity limit on a memdup_user(), so it went to 256 and this to 192, + * which leaves the frame's own objects room several times over. */ +#define SGX_FRAME_MAX_TEX 192u + struct sgx_bo *frame_tex[SGX_FRAME_MAX_TEX]; + unsigned nframe_tex; + /* Grown on demand rather than fixed. Sixty-four records was below what + * real content asks for - glxgears alone issues eighty-six draws a + * frame - and the ones past the limit were dropped, so geometry went + * missing with nothing said. What still bounds this is the room the + * records occupy in the vertex buffer, which ctx_emit_draw_records() + * checks against XPSB_VTX_OFF. */ + unsigned range_cap; + int range_open; /* the current record is still taking draws */ + unsigned fs_uploaded; /* instructions written to the USSE slot */ + unsigned fs_ntemps; /* temporaries that slot's code really uses */ + /* Where the fragment program was put when it did not fit the inline + * slot, and how long it is - what the relocation names it by. Zero + * means it is at the captured offset and the frame's own relocation + * is right as it stands. */ + uint32_t fs_use_off, fs_use_size; + /* The data-segment size ctx_sec_pds() laid out for the secondary + * program, which a per-record copy of that program has to carry. */ + unsigned sec_pds_dwords; + /* And the sa registers that program's one transfer loads. It is built + * from record 0's program and every record copies it, so a record + * whose own program reaches further reads the bank's leftovers. */ + unsigned sec_sa_dwords; + struct sgx_vertex_buffer vb[SGX_MAX_VERTEX_ELEMENTS]; + unsigned nvb; + struct sgx_vertex_element ve[SGX_MAX_VERTEX_ELEMENTS]; + unsigned nve; + int hwtcl; /* the transform runs on the USSE */ + unsigned vtx_floats; /* the record's stride, 11 or 14 */ + /* one record staged in ordinary memory before it is written */ + float *vtx_row; + unsigned vtx_row_cap; + /* An indexed draw's own indices, for a caller that has already + * de-duplicated its vertices. NULL means the upload writes the + * sequential ones it always did. Consumed by the next sgx_draw() and + * cleared by it, so it cannot leak into the draw after. */ + const uint16_t *draw_idx; + unsigned draw_nidx; + unsigned pri_pds_dwords; + /* Where the frame's primary PDS program was built, when that is not + * the captured slot. The per-record copies are taken from it. */ + unsigned pri_pds_off;/* size of the primary program the issues built */ + unsigned prim_dropped; /* primitives with no path on this part */ + float line_width; /* what a widened line segment is drawn at */ + float point_size; /* and what a widened point is drawn at */ + /* The object the frame's primitives are, as SGX_ISP_OBJ_*. Triangles + * unless the ISP is rasterising points or lines itself, in which case + * the draw's indices are handed over one or two at a time instead of + * being widened into triangles on the CPU. One frame is one type: a + * change of primitive re-emits the ISP state, so the frame splits. */ + unsigned prim_objtype; + /* Set when the record carries a point size for the MTE to read. A + * sprite takes its size from the vertex, not from any ISP field - + * measured: the width at [31:28] leaves a native point rasterising + * nothing at 1, 8 and 16 alike - so the size rides at the end of the + * record and bit 8 of the output selects says it is there. */ + int point_size_in_record; + float vs_const[SGX_HWTCL_MAX_UNI_LIMM]; /* its uniforms, as the program reads them */ + unsigned vs_const_dwords; + int vs_fixed; /* run the built-in transform, not the caller's */ + unsigned vtx_uploaded; /* vertices written for the last draw */ + int target_is_scanout; /* the display owns it; do not allocate */ + unsigned dumped_cmd; /* SGX_DUMP_CMD frame counter */ + struct sgx_bo *target_bo; /* the caller's, if it owns the target */ + /* The caller's depth attachment, when a framebuffer has one that is + * a resource rather than the context's own buffer: rendering into a + * depth texture is what a shadow map is, and without this the render + * wrote the context's own buffer while the sample read the caller's, + * which had never been written at all. */ + struct sgx_bo *depth_bo; + uint64_t depth_prev_va; + uint64_t target_prev_va; /* where it was before it became one */ + /* The caller's textures, one per unit the frame relocates: each is + * rebound at the address that unit's relocation names. */ + struct sgx_bo *tex_bo[SGX_MAX_TEX_UNITS]; + uint64_t tex_prev_va[SGX_MAX_TEX_UNITS]; + uint32_t clear_color; + unsigned logicop, logicop_func; /* GL_COLOR_LOGIC_OP, when enabled */ + enum sgx_blend_op blend_op; /* SGX_BLEND_NONE for no blending */ + struct sgx_blend_desc blend_desc; /* what the factors asked for */ + int blend_insn; /* and whether it can be built */ + uint32_t blend_const; /* glBlendColor, packed ARGB8888 */ + int blend_on; + int blend_translucent; + int clear_on; /* the frame begins by clearing colour and depth */ + int zclear_on; /* ... or depth alone, for a depth-only clear */ + uint32_t clear_depth; /* glClearDepth, as float bits */ + unsigned clear_stencil; /* glClearStencil, eight bits */ + /* Whether the depth buffer holds every tile of the picture, which is + * what lets a continuation pass load it back (ZLS). Only a clearing pass + * establishes it: its background object covers every region, so every + * region is stored. A pass that merely continues keeps it - the tiles + * it touches are re-stored and the ones it leaves empty are skipped, + * so what memory holds stays whole either way. Without the load-back a + * frame split restarts depth at the far plane, which is ioquake3's + * dynamic lights shining through walls. */ + int depth_stored; + /* A clear with no draw behind it is still a frame that has to reach + * the hardware. Without this a glClear that nothing follows is simply + * lost, which is most of what a display server asks for. */ + int clear_pending; + /* The fragment stage's uniforms. They live at sa[0] upward, below the + * sampler blocks and the literal pool, and nothing uploaded them - so + * a shader whose colour is a uniform painted with whatever the bank + * happened to hold. */ + uint32_t fs_const[SGX_MAX_FS_CONST_DW]; + /* One snapshot per distinct set the frame uses, so a record renders + * with the uniforms that were bound when it opened rather than with + * whatever the last draw of the frame left behind. */ + uint32_t (*fs_uni_set)[SGX_MAX_FS_CONST_DW]; + unsigned *fs_uni_set_dwords; + unsigned nfs_uni_set, fs_uni_set_cap; + unsigned fs_uni_set_last; /* the set the draw before matched */ + unsigned fs_const_dwords; + unsigned scanout_pitch, scanout_w, scanout_h; + uint32_t *cmd; /* the four streams, host side */ + int cookie_valid; + + /* The bound programs. A draw without both is refused: the hardware has + * no fixed-function fallback, so there is nothing to fall back to. */ + const struct sgx_shader *vs, *fs; + + /* Sampler units. A view and a state are separate objects in Gallium and + * separate here, but the hardware has one pair of words per unit, so + * they are combined when the frame is built rather than at bind. */ + struct sgx_sampler_view views[SGX_MAX_SAMPLERS]; + struct sgx_sampler_state samplers[SGX_MAX_SAMPLERS]; + unsigned view_bound; /* bitmask of units with a view */ + unsigned nviews; /* highest bound unit, plus one */ + + /* The secondary attribute bank the bound programs need filled before + * they run, recomputed on bind. A flush uploads it. */ + unsigned sa_dwords; + unsigned uploads; /* pools uploaded, for the test to count */ + unsigned tex_emitted; /* sampler word pairs written */ + + unsigned draws; /* draws accumulated since the last flush */ + unsigned flushes; + int last_fence; /* out fence of the most recent flush, or -1 */ +}; + +int sgx_context_init(struct sgx_context *ctx, struct sgx_winsys *ws); +void sgx_context_fini(struct sgx_context *ctx); + +/* pipe_context::set_framebuffer_state. Sizes the scene and allocates whatever + * the new size needs; a resize is what makes the parameter heap grow. */ +int sgx_set_framebuffer(struct sgx_context *ctx, + const struct sgx_framebuffer *fb); + +/* Count the records that follow into one of the ISP's eight visibility + * counters, or stop counting with -1. Opens a record of its own, so the draw + * before the query is not in it. Returns 0, or -EINVAL for an index outside + * the three bits ISP word B has for it. */ +int sgx_set_vistest(struct sgx_context *ctx, int reg); +/* Whether any record in the frame about to be submitted is armed, which is + * what tells the kernel to harvest the counters when the render ends. */ +int sgx_frame_has_vistest(const struct sgx_context *ctx); + +/* pipe_context::bind_depth_stencil_alpha_state and ::bind_rasterizer_state */ +void sgx_bind_dsa(struct sgx_context *ctx, const struct sgx_dsa_state *dsa); +void sgx_bind_rasterizer(struct sgx_context *ctx, + const struct sgx_rasterizer_state *r); + +/* pipe_context::bind_blend_state. The operator replaces the fragment program + * with the two SOP2 instructions that carry it out, so a blended draw runs the + * hardware's compositing shader rather than the caller's - which is right for + * a compositor drawing a texture over what is there, and is why anything else + * is refused at translation rather than blended wrongly. */ +void sgx_bind_blend(struct sgx_context *ctx, const struct sgx_blend_state *b); +/* pipe_context::set_blend_color, packed by sgx_blend_const_pack(). The + * constant-colour factors load it into the program with the blend. */ +void sgx_set_blend_color(struct sgx_context *ctx, uint32_t packed); + +/* pipe_context::bind_vs_state and ::bind_fs_state. A shader that did not + * compile is refused here rather than at draw time, because by draw time the + * caller has no way to know which of the two was bad. */ +/* The scene the parameter heap is sized for. Set before the framebuffer, which + * is where the heap is allocated. */ +/* Render into the console framebuffer itself. The target is then the display's + * memory, mapped by the kernel at the same window an allocated one would use, + * so the frame becomes visible without a copy. Call before the framebuffer is + * set: it decides the size. */ +int sgx_use_scanout(struct sgx_context *ctx); + +/* Render into an object the caller owns rather than one the context allocates + * - the colour buffer a pipe_resource stands for. The frame's relocations put + * the target at one address, so binding one there unbinds whatever was there + * before; only the bound framebuffer's colour buffer is at it. Call before the + * framebuffer is set, which is what allocates around it. */ +int sgx_use_target(struct sgx_context *ctx, struct sgx_bo *bo); +/* Drop the caller's colour target, so the next framebuffer gets one of the + * context's own at the size it asks for. */ +int sgx_release_target(struct sgx_context *ctx); + +/* Render depth into the caller's object rather than the context's own. need + * is what the frame will write - the padded target size - and an object + * smaller than that is refused, because the raster pass writes past the + * framebuffer's last row. */ +int sgx_use_depth(struct sgx_context *ctx, struct sgx_bo *bo, uint64_t need); +int sgx_release_depth(struct sgx_context *ctx); + +/* Sample from an object the caller owns. The frame relocates the texture + * address out of its own object, so a caller's texture has to be at that + * address to be the one sampled - writing the address into the descriptor is + * not enough, the relocation puts the frame's back. Only unit 0 has an address + * of its own so far; a second unit needs the two-unit frame's. */ +int sgx_use_texture(struct sgx_context *ctx, unsigned unit, struct sgx_bo *bo); + +/* Put the context's own texture back behind a unit's fixed address when a + * caller's object leaves it, so the frame never samples an unmapped address. */ +int sgx_release_texture(struct sgx_context *ctx, unsigned unit); +/* The kernel takes this many buffers in one submit. */ +#define SGX_SUBMIT_MAX_BO 256u +int sgx_frame_add_tex(struct sgx_context *ctx, struct sgx_bo *bo); +void sgx_frame_drop_tex(struct sgx_context *ctx, const struct sgx_bo *bo); +int sgx_refresh_texture(struct sgx_context *ctx, struct sgx_bo *bo); +int sgx_note_texture(struct sgx_context *ctx, unsigned unit, + struct sgx_bo *bo); +int sgx_context_frame_uses(const struct sgx_context *ctx, + const struct sgx_bo *bo); + +/* The colour the frame's clearing draws write. Only the low 21 bits of the + * packed pixel reach the hardware: the clear is a USSE program whose immediate + * is that wide, so the red channel keeps five of its eight bits. Black, and + * anything dark, is exact. */ +int sgx_set_clear_color(struct sgx_context *ctx, uint32_t packed); +/* The depth the clearing records write and the background object's depth + * (EUR_CR_ISP_BGOBJDEPTH), and the stencil the background object carries + * (EUR_CR_ISP_BGOBJ [7:0]) - the values a frame without a depth load + * begins from. */ +int sgx_set_clear_depth(struct sgx_context *ctx, float depth); +int sgx_set_clear_stencil(struct sgx_context *ctx, unsigned stencil); +/* The five-bit temporary-register field's value for a program of this size, + * or zero if it does not fit. */ +unsigned sgx_temp_field(unsigned ntemps); +unsigned sgx_temp_field_hi(unsigned ntemps); +int sgx_temp_ok(unsigned ntemps); +unsigned sgx_temp_max(void); +unsigned sgx_sa_max(void); + +/* Where a fragment program's constants sit in the heap before the secondary + * PDS program moves them into the secondary attribute bank. Nothing else in + * the heap or the relocations names this block. */ + +/* Ask for the frame's clear to be submitted even if no draw joins it. */ +void sgx_request_clear(struct sgx_context *ctx); +/* The fragment stage's uniform block, as the caller set it. */ +int sgx_set_fs_constants(struct sgx_context *ctx, const uint32_t *v, + unsigned n); +/* Whether a clear is waiting for a frame to carry it. */ +int sgx_clear_pending(const struct sgx_context *ctx); + +/* Whether the frame clears before it draws. + * + * Dropping the two clearing draws does not preserve the destination: measured + * on hardware, a second frame drawn with them dropped leaves the target black + * everywhere, including where the first frame had drawn. That is what a + * deferred renderer does - it resolves every tile it processes, whether any + * geometry reached it or not - so preserving the destination is a matter of + * loading it into the tile buffer rather than of not clearing it, and how to + * ask for that is not decoded. Off is therefore not usable yet, and the frame + * always clears. */ +int sgx_set_clear_enable(struct sgx_context *ctx, int on); +/* Clear depth while the colour survives. Turning the colour clear off turns + * this off with it: both end when the frame's clear does. */ +int sgx_set_zclear_enable(struct sgx_context *ctx, int on); + +int sgx_set_max_prims(struct sgx_context *ctx, unsigned prims); + +/* Move the vertex transform onto the USSE. Has to be set before the + * framebuffer, because it changes what the frame is built from. */ +int sgx_set_hwtcl(struct sgx_context *ctx, int on); + +/* The vertex program's uniforms. Under hardware TCL these are DMAed into the + * primary attribute bank just past the vertex record, so there are at most + * eighteen dwords of them. */ +int sgx_set_vs_constants(struct sgx_context *ctx, const float *v, unsigned n); + +/* Run the built-in transform-only program rather than the bound one. It reads + * a row-major matrix from the uniforms and emits window coordinates, which is + * the arrangement tools/baremetal/sgxtri.c proves on this part - so this is + * what says whether a frame that draws nothing is the frame's fault or the + * compiled program's. */ +int sgx_set_vs_fixed(struct sgx_context *ctx, int on); + +/* The transform hardware TCL applies when the caller's own vertex program is + * not available: sixteen dwords, row major. */ +int sgx_set_vs_matrix(struct sgx_context *ctx, const float *m); + +int sgx_set_vertex_buffers(struct sgx_context *ctx, + const struct sgx_vertex_buffer *vb, unsigned n); +int sgx_set_vertex_elements(struct sgx_context *ctx, + const struct sgx_vertex_element *ve, unsigned n); + +int sgx_bind_vs(struct sgx_context *ctx, const struct sgx_shader *vs); + +/* The frame's own vertex program, which is what the hardware runs whatever the + * caller bound: a caller's vertex shader is executed on the CPU, so what the + * part sees is already in screen space and needs nothing but a pass-through. + * A draw is refused without a vertex program bound, and this is the one. */ +const struct sgx_shader *sgx_passthrough_vs(void); + +/* Which of the bound fragment program's inputs are texture coordinates. The + * record has one coordinate slot and one colour slot, so whoever fills the + * record has to know which varying belongs in which. */ +unsigned sgx_fs_tex_inputs(const struct sgx_context *ctx); +unsigned sgx_fs_inputs(const struct sgx_context *ctx, + const unsigned char **sem, const unsigned char **idx); +int sgx_fs_sets_swapped(const struct sgx_context *ctx); +int sgx_fs_set_varying(const struct sgx_context *ctx, unsigned set); +int sgx_fs_packed_in(const struct sgx_context *ctx); +int sgx_fs_set_coord(const struct sgx_context *ctx, unsigned set, + unsigned char *c0, unsigned char *c1); +int sgx_fs_coord_split(const struct sgx_context *ctx, int *set, + unsigned char *c); +void sgx_dump_target(struct sgx_context *ctx, const char *path); +/* Diagnostic sink: SGX_LOG names a file, else stderr. The X server closes + * the driver's stderr, so under X a file is the only channel. */ +FILE *sgx_log(void); +int sgx_fs_fragcoord_set(const struct sgx_context *ctx); +/* The depth of the volume bound to the unit that samples a coordinate set, + * 0 when that unit's view is not a volume. */ +unsigned sgx_fs_set_depth(const struct sgx_context *ctx, unsigned set); +const struct xpsb_attribs *sgx_fs_attribs(const struct sgx_context *ctx); +const struct sgx_shader *sgx_ctx_fs(const struct sgx_context *ctx); +unsigned sgx_fs_vary_floats(const struct sgx_context *ctx); +void sgx_update_vtx_floats(struct sgx_context *ctx); +int sgx_scissor_clips(const struct sgx_context *ctx); +int sgx_mix_prim(void); +unsigned sgx_record_extra_dwords(const struct sgx_context *ctx); +/* Whether the frame prepends the ISP's own coordinate set - the UV a point + * object produces - ahead of the program's sets. */ +int sgx_sprite_uv_set(const struct sgx_context *ctx); +int sgx_bind_fs(struct sgx_context *ctx, const struct sgx_shader *fs); + +/* pipe_context::set_sampler_views and ::bind_sampler_states. A NULL view + * unbinds the unit. The view is validated here rather than at draw, so a + * texture the descriptor cannot express is reported while the caller still + * knows which one it was. */ +int sgx_set_sampler_view(struct sgx_context *ctx, unsigned unit, + const struct sgx_sampler_view *v); +int sgx_bind_sampler_state(struct sgx_context *ctx, unsigned unit, + const struct sgx_sampler_state *st); + +/* pipe_context::draw_vbo, reduced to what the hardware needs to know */ +int sgx_draw(struct sgx_context *ctx, unsigned vertex_count); +/* Triangle corners the frame can still take, a whole triangle at a time. + * Zero means it has to be sent first. */ +unsigned sgx_draw_room(const struct sgx_context *ctx); + +/* How many draw records the pending frame holds. */ +unsigned sgx_context_ranges(const struct sgx_context *ctx); + +/* Draws that found no free record. A draw whose state matches the open record + * joins it, so this counts only genuinely distinct ones past the limit. */ +unsigned sgx_context_dropped(const struct sgx_context *ctx); +/* Call once per swap. Under SGX_PERF it reports submits per swap, which is + * what frame splitting costs. */ +void sgx_perf_swap(void); +/* Name the split that is about to end the frame, for that report. */ +void sgx_perf_note(const char *why); +/* Wall-clock accounting for the same report, because splits and frame rate are + * not proportional here and a count cannot say where the time went. mark() + * returns 0 when SGX_PERF is off and add() with 0 does nothing, so an + * unmeasured section costs a branch. which: 0 the frame build, 1 the idle + * wait, 2 the flush. */ +uint64_t sgx_perf_mark(void); +void sgx_perf_add(int which, uint64_t t0); +void sgx_context_dump_ranges(const struct sgx_context *ctx); +void sgx_context_drop_frame(struct sgx_context *ctx); +void sgx_forget_fs(struct sgx_context *ctx, const void *fs); +void sgx_forget_vs(struct sgx_context *ctx, const void *vs); + +/* Whether the uniforms bound now differ from what the open draw record was + * opened with, so that the caller can start a new one. */ +int sgx_vs_constants_changed(const struct sgx_context *ctx); + +/* End the draw record the next draw would have joined. Consecutive draws share + * one record - a record is a primitive block and the tiler allocates parameter + * memory per block, so one per draw is both slower and far hungrier: ninety-two + * of them, one per strip of the three gears, exhausted the heap outright - and + * this is how the layer above says the next draw needs its own. */ +void sgx_next_draw_record(struct sgx_context *ctx); + +/* pipe_context::flush. Builds the streams and submits them. */ +int sgx_flush(struct sgx_context *ctx); + +/* Whether the MTE applies the viewport rather than a shader epilogue. */ +int sgx_mte_viewport(void); + +/* The caller's viewport, for the MTE's group 8: window = scale * ndc + + * translate, which is the order the group carries them in. */ +int sgx_set_viewport(struct sgx_context *ctx, const float *scale, + const float *translate); + +#endif diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_cookie.h mesa-26.2.2/src/gallium/drivers/sgx/sgx_cookie.h --- mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_cookie.h 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/sgx_cookie.h 2026-09-08 10:57:36.681299320 +0200 @@ -0,0 +1,187 @@ +/* SGX scene cookie - the macro-tile geometry and heap layout a fire needs. + * + * A transcription of xpsb_scene_info() and xpsb_ta_mem_info() in + * tools/xpsb-open/xpsb_hw.c, from Xpsb_scene_info at 0x3a40 and + * Xpsb_ta_mem_info at 0x3d70. + * + * Unlike sgx_kick.h and sgx_scene.h this touches no register at all: it is + * arithmetic on the render target size, producing the sixteen dwords that + * sgx_scene_switch_fire() consumes. That makes it exhaustively testable rather + * than sampled, and test/sgx_cookie_test.c does sweep it exhaustively over + * every render target size the hardware accepts. + * + * The two workaround flags are per core revision. SGX535 rev 1.2.1 - the only + * part this project has - enables neither, so those paths are transcribed from + * the binary and swept against it but have never run on silicon. They are + * marked where they occur. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: GPL-2.0-only + */ +#ifndef _SGX_COOKIE_H_ +#define _SGX_COOKIE_H_ + +#define SGX_C_ALIGN4(v) (((v) + 3u) & ~3u) + +/* The scene cookie is sixteen dwords. Twelve come from here; cookie[13], + * cookie[14] and cookie[15] are the out-of-memory state, owned by the abort + * handler and the fire path, and are zeroed at scene setup. */ +#define SGX_C_COOKIE_DWORDS 16 + +struct sgx_cookie_caps { + int large_scene; /* vopt 18: >2047 scenes keep 2x2 macro tiles */ + int wide_mtile; /* vopt 20: 2 tiles per macro tile, not 1/2 */ +}; + +static inline unsigned int sgx_roundup_pow2(unsigned int v) +{ + unsigned int m, p; + + if (!v) + return 0; + m = v & 0x0fffffffu; + p = 1; + while (m > p) + p += p; + return p; +} + +/* A macro tile is a multiple of four tiles a side at one sample, and of two + * when that axis carries two samples: SetMT4Mode() and SetMT16Mode() in + * sgxrender_targets.c:587-666 pick the granularity from + * ui16MSAASamplesInX/Y. The vendor binary this file transcribes only ever + * built one-sample scenes, so the two-sample arm comes from the DDK. */ +#define SGX_C_ALIGN_MT(v, s) ((s) == 1u ? SGX_C_ALIGN4(v) : \ + (((v) + 1u) & ~1u)) + +/* Xpsb_scene_info at 0x3a40, with the sample count the vendor binary had no + * argument for. sx and sy are samples along each axis, 1 or 2: the part + * reaches 1x1 and 2x2 and nothing else, and refusing anything here is what + * keeps a scene from being described at a size the tiler will overrun. + * + * What is in sample tiles and what is in pixel tiles: + * + * cookie[2], cookie[3], cookie[4] are the macro-tile registers, which stay + * in pixel tiles - the part scales them itself from EUR_CR_TE_AA + * (sgxkick_client.c:1083-1089 hands EUR_CR_TE_MTILE1/2 the unmultiplied + * values while :1067-1077 sets the AA bits). + * + * cookie[5] and cookie[6] are the screen and pixel extents, also pixel + * (EUR_CR_TE_SCREEN and EUR_CR_MTE_SCREEN, sgxkick_client.c:1090-1105). + * + * The tail-pointer array and the region-header array are one entry per + * *sample* tile, so both grow by sx*sy: sgxrender_targets.c:2408-2409 + * multiplies tiles-per-macrotile by the sample counts before :2433-2435 + * sizes the region array and :2465-2468 sizes the tail pointers. + * + * Returns the parameter-buffer size the scene needs and the page range the + * caller has to clear before the tiler runs. */ +static inline int sgx_scene_info_ms(const struct sgx_cookie_caps *caps, + unsigned int w, unsigned int h, + unsigned int sx, unsigned int sy, + unsigned int *cookie, unsigned int *size, + unsigned int *clear_p_start, + unsigned int *clear_num_pages) +{ + unsigned int tx = (w + 15u) >> 4, ty = (h + 15u) >> 4; + unsigned int mx, my, ex, ey, upt, px, py, pow2, base; + + if ((sx != 1u && sx != 2u) || (sy != 1u && sy != 2u)) + return -22; /* -EINVAL */ + mx = SGX_C_ALIGN_MT((tx + 1u) >> 1, sx); + my = SGX_C_ALIGN_MT((ty + 1u) >> 1, sy); + + if ((w > 0x7ff || h > 0x7ff) && !caps->large_scene) { + /* 4x4 macro tiles. Reachable on this part - a render target + * wider or taller than 2047 takes it. */ + cookie[0] = 0x80000000; + upt = 4; + mx = SGX_C_ALIGN_MT((tx + 3u) >> 2, sx); + my = SGX_C_ALIGN_MT((ty + 3u) >> 2, sy); + ex = mx; + ey = my; + cookie[2] = 0x80000000u | (mx << 22) | (mx << 13) | (mx * 3u); + cookie[3] = (my << 22) | (my << 13) | (my * 3u); + } else { + cookie[0] = 0; + upt = 2; + if (caps->wide_mtile) { + /* never taken on SGX535 rev 1.2.1 */ + ex = SGX_C_ALIGN4(tx * 2u); + ey = SGX_C_ALIGN4(ty * 2u); + if (ex > 0xff) + ex = 0xff; + if (ey > 0xff) + ey = 0xff; + } else { + ex = mx; + ey = my; + } + cookie[2] = (ex << 22) | (ex << 12) | ex; + cookie[3] = (ey << 22) | (ey << 12) | ey; + } + + cookie[1] = mx * my; + cookie[4] = ex * ey; + cookie[5] = ((ty - 1u) << 12) | (tx - 1u); + cookie[6] = ((h - 1u) << 12) | (w - 1u); + cookie[8] = 0; + + px = upt * mx * sx; + py = upt * my * sy; + pow2 = sgx_roundup_pow2(px); + if (sgx_roundup_pow2(py) > pow2) + pow2 = sgx_roundup_pow2(py); + cookie[7] = ((pow2 * pow2 * 4u) + 0xfffu) & ~0xfffu; + + base = cookie[7] + px * py * 12u; + cookie[9] = base; + cookie[10] = base + 0x50; + cookie[11] = base + 0x50 + 0x90; + cookie[12] = 0; + cookie[13] = 0; + cookie[14] = 0; + + *size = base + 0x50 + 0x90 + 0x40; + *clear_p_start = cookie[8] >> 12; + *clear_num_pages = ((cookie[7] + 0xfffu) >> 12) - *clear_p_start; + return 0; +} + +/* One sample a pixel, which is what everything but a multisampled render + * asks for. */ +static inline int sgx_scene_info(const struct sgx_cookie_caps *caps, + unsigned int w, unsigned int h, + unsigned int *cookie, unsigned int *size, + unsigned int *clear_p_start, + unsigned int *clear_num_pages) +{ + return sgx_scene_info_ms(caps, w, h, 1u, 1u, cookie, size, + clear_p_start, clear_num_pages); +} + +/* Xpsb_ta_mem_info at 0x3d70. This only validates: psb discards the size and + * allocates its own parameter heap, so the responder sets a floor rather than + * choosing the heap, and an undersized request is the one thing it refuses. */ +static inline int sgx_ta_mem_info(unsigned int pages, unsigned int *cookie, + unsigned int *size) +{ + unsigned int avail, i; + + *size = 0x620000; + if (pages <= 0x61f) + return -12; /* -ENOMEM */ + + avail = pages - 0x600; + cookie[2] = pages; + cookie[3] = avail; + cookie[4] = avail; + cookie[5] = avail; + cookie[6] = (avail >= 0x41) ? (pages - 0x640) : 0; + for (i = 7; i < 13; i++) + cookie[i] = 0; + return 0; +} + +#endif diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_drm.h mesa-26.2.2/src/gallium/drivers/sgx/sgx_drm.h --- mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_drm.h 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/sgx_drm.h 2026-09-08 10:57:36.681331863 +0200 @@ -0,0 +1,373 @@ +/* SPDX-License-Identifier: MIT */ +/* + * UAPI for a PowerVR SGX535 render node on Poulsbo, as a render side of the + * existing gma500 display driver. + * + * This is a proposal, not a shipped interface. It is written against what the + * hardware was measured to need - see MESA-DRI-GAPS.md and GL-IMPLEMENTATION.md + * - and deliberately against what the 2008 psb driver did, in three places: + * + * - No xhw round trip. psb pushed scene management into userspace and had + * the kernel ask it questions over a mailbox. The responder is open C now + * (tools/xpsb-open/xpsb_hw.c, ~900 lines), so it belongs in the kernel and + * userspace never sees a scene cookie. + * + * - No relocations. A frame currently applies 77 TA and 16 raster + * relocations; userspace assigns GPU addresses itself here and the kernel + * only validates that a named object is mapped where the stream says. + * Relocation-based submit is an argument upstream would not accept. + * + * - Real fences. dma_fence out of SUBMIT rather than the sequence numbers + * and psb_fence_wait of the old driver. + * + * Copyright (C) 2026 RenĂ© Rebe + */ +#ifndef _UAPI_SGX_DRM_H_ +#define _UAPI_SGX_DRM_H_ + +/* One header, two contexts: the kernel has the real , the host + * tests have a stub. There used to be two copies of this file for that reason + * and they had already drifted - SGX_VM_SCENE existed in one and not the + * other. kernel/sgx_drm.h is a symlink to this now, so they cannot. */ +#ifdef __KERNEL__ +#include +#else +#include "drm.h" +#endif + +#if defined(__cplusplus) +extern "C" { +#endif + +#define DRM_SGX_GEM_NEW 0x00 +#define DRM_SGX_GEM_MAP 0x01 +#define DRM_SGX_GEM_WAIT 0x02 +#define DRM_SGX_VM_BIND 0x03 +#define DRM_SGX_SUBMIT 0x04 +#define DRM_SGX_GET_PARAM 0x05 +#define DRM_SGX_USE_BASE 0x06 +#define DRM_SGX_SCANOUT 0x07 +#define DRM_SGX_WAIT 0x08 +#define DRM_SGX_BLIT 0x09 +#define DRM_SGX_VISTEST 0x0a + +/* + * A buffer object. Nothing here is SGX specific except the flags: the display + * side of gma500 already has GEM, and a render node has to share it so a + * scanout buffer can be rendered into without a copy. + */ +struct drm_sgx_gem_new { + __u64 size; + __u32 flags; +#define SGX_BO_SCANOUT (1 << 0) /* must satisfy the display stride */ +#define SGX_BO_CPU_CACHED (1 << 1) /* the parameter heap is uncached */ + __u32 handle; /* out */ +}; + +struct drm_sgx_gem_map { + __u32 handle; + __u32 pad; + __u64 offset; /* out: mmap cookie */ +}; + +struct drm_sgx_gem_wait { + __u32 handle; + __u32 flags; + __s64 timeout_ns; +}; + +/* + * Explicit address binding. Userspace picks the GPU address, which is what + * lets the submit path be relocation free; the kernel programs the SGX MMU + * (gma500 already has mmu.c) and refuses overlaps. + * + * The address space is not flat: the SGX takes its parameter heap, its render + * targets and its PDS window through different requestors with different base + * registers, so a binding names which one it is for. + */ +struct drm_sgx_vm_bind { + __u32 handle; + __u32 flags; +#define SGX_BIND_UNBIND (1 << 0) + __u64 gpu_va; + __u64 offset; + __u64 size; + __u32 window; +#define SGX_VM_PARAM 0 /* the DPM parameter heap */ +#define SGX_VM_PDS 1 /* heap, USSE and PDS programs */ +#define SGX_VM_SURFACE 2 /* render targets, depth, textures */ +#define SGX_VM_RASTGEOM 3 +#define SGX_VM_SCENE 4 /* the scene and the DPM's own tables */ +#define SGX_VM_COUNT 5 /* windows, so nothing hardcodes four */ + __u32 pad; +}; + +/* + * One frame. + * + * The two register streams are what the hardware actually consumes: a TA + * stream that binds the vertex source and starts the tiler, and a raster + * stream that runs the 3D core over what the tiler binned. They are arrays of + * {register offset, value} pairs, the same shape psb_reg_submit() took, and + * the kernel validates every offset against a whitelist rather than trusting + * them - that check does not exist in the old driver and is the reason its + * submit path could not be upstreamed. + * + * There is no oom_cmds member. The out-of-memory recovery is the kernel's + * business now: it owns the scene, and work/heap-probe plus the recovery in + * xpsb_hw.c showed that getting it wrong loses geometry silently. Userspace + * cannot be trusted with a path whose failure mode is a slightly wrong frame. + */ +struct drm_sgx_submit { + __u64 ta_stream; /* __u32 pairs */ + __u64 raster_stream; + __u32 ta_stream_count; /* in dwords, so 2 per pair */ + __u32 raster_stream_count; + + /* The out-of-memory command stream. The fire path branches on whether + * it was given one - scene_fire_common() sets the cookie's bit 31 only + * when num_oom_cmds is non-zero - so this is not optional decoration + * for a frame that never runs out: without it the DPM does not recycle + * parameter pages and an animation walks off the end of the heap. */ + __u64 oom_stream; + __u32 oom_stream_count; + __u32 oom_pad; + + __u64 bo_handles; /* every object the frame touches */ + __u32 bo_count; + __u32 flags; +/* The most handles one submit may name. A sanity bound rather than a hardware + * one: the kernel copies the list with memdup_user() and nothing downstream is + * sized by it. It was 64, which put the driver's texture budget at 48 and cost + * ioquake3 a whole frame per swap when a scene sampled more than that. */ +#define SGX_SUBMIT_MAX_HANDLES 256u +#define SGX_SUBMIT_NO_PRESENT (1 << 0) +/* Ask for out_sync_fd. Without this no descriptor is created: the fire is + * synchronous and the fence is signalled before it is handed out, so a caller + * that does not wait on one was only leaking it. */ +#define SGX_SUBMIT_FENCE_OUT (1 << 1) +/* Come back once the render is fired rather than once it has ended, so the + * caller can build the next frame while this one is on the core. Nothing about + * the frame changes; what changes is who waits for it and when. + * + * The wait is not skipped, it is moved: the next submit drains it before it + * reprograms anything, and DRM_IOCTL_SGX_WAIT is where userspace drains it + * before it reads a surface back or overwrites state the render is still + * reading. A caller that sets this and then writes into its heap, its shader + * code or a bound texture without waiting is racing the core, and the + * corruption that follows is its own. */ +#define SGX_SUBMIT_DEFER_WAIT (1 << 2) +/* This frame has objects armed for the ISP's visibility counters, so the + * kernel has to harvest them when the render ends. The vendor's microkernel + * takes the same flag through the kick - SGXMKIF_RENDERFLAGS_GETVISRESULTS, + * "setting this flag will cause the uKernel to collect the visibility + * results" (services4/include/sgx_mkif_client.h:54-57), set from + * SGX_KICKTA_FLAGS_GETVISRESULTS in sgxkick_client.c:2338-2354 - and only + * then runs the accumulate-and-clear at 3d.asm:714-776. A frame without it + * arms no object, so the counters do not move and there is nothing to read. + */ +#define SGX_SUBMIT_VISTEST (1 << 3) + +/* Two samples a pixel in each axis - four in all, the only multisample mode + * this core has. SGX_FEATURE_MSAA_2X_IN_X and SGX_FEATURE_MSAA_2X_IN_Y are + * what would make a two-sample render target legal and the SGX535 block of + * the vendor's sgxfeaturedefs.h defines neither, so there is no flag for one. + * + * The kernel needs this because it, not userspace, builds the scene: the + * tail-pointer and region-header arrays hold one entry per sample tile, so a + * multisampled scene needs four times as many of each. A frame that programs + * EUR_CR_TE_AA without this flag would have the tiler write past the end of + * the region array. */ +#define SGX_SUBMIT_MSAA_2X2 (1 << 4) + + /* The render target's geometry, which the kernel needs to size the + * scene and the parameter heap - xpsb_scene_info() and + * dpm_param_pages(). Userspace does not get to pick the heap size: + * that estimate is what the recovery path depends on. */ + __u32 width, height; + + /* Must be -1. No path here waits on a fence before it fires, so a + * submit carrying one is refused rather than run out of the order the + * caller asked for. */ + __s32 in_sync_fd; + /* Out, and only when SGX_SUBMIT_FENCE_OUT asked for one; -1 otherwise. + */ + __s32 out_sync_fd; +}; + +/* + * USE base registers. + * + * A shader is fetched through one of thirteen base registers, and a PDS + * program names the register and an offset within it rather than an address. + * Those registers are a global hardware resource: a client that could write + * them could redirect another client's code fetch, so they are not in the + * submit whitelist and userspace cannot set them. + * + * Instead it asks. Given the address and size of some code and which data + * master will run it, the kernel allocates or reuses a base register and + * returns the register number and the offset within it - which is exactly what + * a PDS program needs to encode. The assignment lasts for the file. + */ +struct drm_sgx_use_base { + __u64 gpu_va; /* where the code is */ + __u32 size; + __u32 data_master; +/* The hardware's own encoding, which is not the obvious order: psb_regman.c + * defaults every managed register to PIXEL by writing 1 into the field. The + * kernel passes this value straight through to CR_USE_CODE_BASE, so naming + * PIXEL 0 here would have selected VERTEX on the device. */ +#define SGX_USE_DM_VERTEX 0 +#define SGX_USE_DM_PIXEL 1 + __u32 reg; /* out: 3..15 */ + __u32 offset; /* out: gpu_va - the register's base */ +}; + +/* + * The console framebuffer, as a render target. + * + * The 3D core renders into the memory the display is already scanning out, so + * a frame becomes visible without anything copying it. The kernel maps that + * memory into the caller's address space at the render-target window and says + * where and how big it is; there is no handle, because the buffer is the + * driver's and userspace never gets to own it. + */ +struct drm_sgx_scanout { + __u64 gpu_va; /* out: where it was mapped */ + __u32 pitch; /* out: bytes per row */ + __u32 width, height; /* out */ + __u32 size; /* out */ + __u32 pad; +}; + +/* + * Drain the device: return once the render the last submit fired has ended. + * + * Only meaningful with SGX_SUBMIT_DEFER_WAIT, and cheap when nothing is + * outstanding. There is no handle: the core runs one frame at a time, so + * "the pending frame" is unambiguous and waiting for it is waiting for all of + * it - the tiler, the render and the parameter-memory dealloc. + */ +struct drm_sgx_wait { + __u32 flags; + __u32 pad; +}; + +/* + * The ISP's visibility counters, which are what an occlusion query counts. + * + * Eight of them, EUR_CR_ISP_VISTEST_VISIBLE0..7 (sgx535defs.h:1378-1415), and + * on this core they are device registers and nothing else: the SGX535 feature + * block does not define SGX_FEATURE_VISTEST_IN_MEMORY (sgxfeaturedefs.h: + * 189-232), so there is no in-memory result the way later cores have. A + * submitted stream is register writes and userspace cannot read a register, so + * the counter has to be read on this side and handed back - which is why this + * is an ioctl of its own rather than a field somewhere. It cannot be a field + * in DRM_IOCTL_SGX_WAIT either: that one is DRM_IOW and giving it an output + * changes its command number, which would break every existing caller. + * + * What comes back is an accumulator per counter, not the register. The kernel + * reads all eight when a render armed with SGX_SUBMIT_VISTEST ends, adds them + * into these, and clears the registers - which is exactly what the vendor's + * microkernel does per render (3d.asm:714-776, the read-add-write into the + * client's eight-dword buffer followed by the all-eight CLEAR write at :767). + * Accumulating is not optional there and is not optional here: on SGX535 the + * client's SGX_KICKTA_FLAGS_DISABLE_ACCUMVISRESULTS is ignored and the flag is + * always set (sgxkick_client.c:2344-2351), and the accumulate runs "after + * every render including SPM partial renders" (3d.asm:1170-1191). A query that + * spans a frame split therefore still counts every part of itself. + * + * The accumulators are per open file and never reset. A caller reads them + * before it arms a query and again after, and the difference is that query's + * count - which is also what lets several queries share one file without a + * reset racing between them. + */ +#define SGX_VISTEST_REGS 8 + +struct drm_sgx_vistest { + __u32 flags; +/* Drain a deferred render before reading, so the counts include the frame the + * last submit fired. Without it the read is a snapshot of what has landed. */ +#define SGX_VISTEST_WAIT (1 << 0) + /* Out: this file still has a render outstanding, so a count that has + * not landed yet is missing from what follows. Always zero when + * SGX_VISTEST_WAIT was asked for and the drain succeeded. */ + __u32 pending; + __u64 count[SGX_VISTEST_REGS]; /* out */ +}; + +/* + * A 2D job: copies and solid fills on the SGX535's 2D block, which is a + * separate engine from the 3D pipeline - no scene, no shaders, no tiler. + * + * The stream is the engine's own command format, block headers as __u32 + * dwords (driver/gma500/sgx_twod.h), and the kernel checks every block before + * it writes one to the slave port: only surfaces, a source offset, blits, + * fences and flushes are accepted, and every surface address - an offset from + * BIF_TWOD_REQ_BASE, which is the surface window's start - has to fall inside + * a binding this file made. A stream is not a register write: nothing in it + * can reach a register, so the whitelist in sgx_check.h does not grow. + * + * Synchronous. The engine runs behind the same lock as the 3D core and shares + * its page-directory register, so a blit drains any render still on the core + * first and has finished when the ioctl returns; the next submit therefore + * sees what it wrote, and it saw what the render before it wrote. A fence is + * handed out already signalled, like the submit's. + */ +struct drm_sgx_blit { + __u64 stream; /* __u32 dwords */ + __u32 stream_count; /* in dwords */ + __u32 flags; +#define SGX_BLIT_FENCE_OUT (1 << 0) +/* Come back once the stream is on the engine rather than once the engine has + * run it, so the caller can carry on while it works. The wait is not skipped, + * it is moved: the kernel drains it wherever it drains a deferred render - at + * the head of the next submit, in VM_BIND before an address can change under + * the engine, in GEM_WAIT and in DRM_IOCTL_SGX_WAIT, which is where userspace + * drains it before reading a surface. A caller that sets this and then reads + * or rebinds without waiting is racing the engine. Refused together with + * SGX_BLIT_FENCE_OUT: that fence is handed out signalled. */ +#define SGX_BLIT_DEFER_WAIT (1 << 1) + __u64 bo_handles; /* every object the stream touches */ + __u32 bo_count; + __u32 pad; + __s32 in_sync_fd; /* must be -1 */ + __s32 out_sync_fd; /* out, with SGX_BLIT_FENCE_OUT */ +}; + +/* + * Parameters. Kept small on purpose: anything userspace can derive, it should. + */ +struct drm_sgx_get_param { + __u32 param; +/* EUR_CR_CORE_REVISION as the hardware reports it: designer in bits 31:24, + * major 23:16, minor 15:8, maintenance 7:0. SGX535 rev 1.2.1 is 0x00010201. */ +#define SGX_PARAM_CORE_ID 0 +#define SGX_PARAM_CORE_CLOCK 1 /* 200 MHz, from the message bus */ +#define SGX_PARAM_VM_START 2 /* per-window, index in value */ +#define SGX_PARAM_VM_SIZE 3 +/* The GPU address BIF_TWOD_REQ_BASE holds; a 2D surface word carries an + * offset from it, 28 bits wide. Refused by a kernel without the 2D path, + * which is how userspace learns it has none. */ +#define SGX_PARAM_TWOD_BASE 4 + __u32 index; + __u64 value; /* in for index, out for value */ +}; + +#define DRM_IOCTL_SGX_GEM_NEW DRM_IOWR(DRM_COMMAND_BASE + DRM_SGX_GEM_NEW, struct drm_sgx_gem_new) +#define DRM_IOCTL_SGX_GEM_MAP DRM_IOWR(DRM_COMMAND_BASE + DRM_SGX_GEM_MAP, struct drm_sgx_gem_map) +#define DRM_IOCTL_SGX_GEM_WAIT DRM_IOW (DRM_COMMAND_BASE + DRM_SGX_GEM_WAIT, struct drm_sgx_gem_wait) +#define DRM_IOCTL_SGX_VM_BIND DRM_IOW (DRM_COMMAND_BASE + DRM_SGX_VM_BIND, struct drm_sgx_vm_bind) +#define DRM_IOCTL_SGX_SUBMIT DRM_IOWR(DRM_COMMAND_BASE + DRM_SGX_SUBMIT, struct drm_sgx_submit) +#define DRM_IOCTL_SGX_GET_PARAM DRM_IOWR(DRM_COMMAND_BASE + DRM_SGX_GET_PARAM, struct drm_sgx_get_param) +#define DRM_IOCTL_SGX_USE_BASE DRM_IOWR(DRM_COMMAND_BASE + DRM_SGX_USE_BASE, struct drm_sgx_use_base) +#define DRM_IOCTL_SGX_SCANOUT DRM_IOWR(DRM_COMMAND_BASE + DRM_SGX_SCANOUT, struct drm_sgx_scanout) +#define DRM_IOCTL_SGX_WAIT DRM_IOW (DRM_COMMAND_BASE + DRM_SGX_WAIT, struct drm_sgx_wait) +#define DRM_IOCTL_SGX_BLIT DRM_IOWR(DRM_COMMAND_BASE + DRM_SGX_BLIT, struct drm_sgx_blit) +#define DRM_IOCTL_SGX_VISTEST DRM_IOWR(DRM_COMMAND_BASE + DRM_SGX_VISTEST, struct drm_sgx_vistest) + +#if defined(__cplusplus) +} +#endif + +#endif /* _UAPI_SGX_DRM_H_ */ diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_hwtcl.c mesa-26.2.2/src/gallium/drivers/sgx/sgx_hwtcl.c --- mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_hwtcl.c 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/sgx_hwtcl.c 2026-09-08 10:57:36.680168417 +0200 @@ -0,0 +1,584 @@ +/* Hardware vertex transform - see sgx_hwtcl.h. + * + * Copyright (C) 2026 RenĂ© Rebe + * SPDX-License-Identifier: MIT + */ +#include +#include +#include +#include + +#include "sgx_hwtcl.h" +#include "xpsb_frame.h" + +/* getenv() is a linear scan of the environment, and these predicates are asked + * once or more per draw. Measured with perf on glmark2's build scene, getenv + * and the strncmp inside it were 16% of the client's CPU, all of it from this + * file. What they answer cannot change during a run, so each site remembers + * it - the same treatment SGX_ENV() gives sgx_context.c, which this file + * cannot include. */ +#define HWTCL_ENV(name) __extension__({ \ + static const char *v_; \ + static int got_; \ + if (!got_) { v_ = getenv(name); got_ = 1; } \ + v_; \ +}) + +/* Heap dwords the vertex PDS program occupies. The captured program is + * mul/movsa/doutu/halt; the DOUTD goes in front of the DOUTU, which pushes + * both it and the HALT down one slot. */ +#define VPDS_SRC_DW 0xeau +#define VPDS_CTL_DW 0xebu +#define VPDS_CODE_DW 0xf4u +#define VPDS_MUL 0x2f030343u +#define VPDS_MOVSA 0x030803e5u +#define PDS_DOUTD_2 0x07040223u /* movs doutd, ds0[2], ds0[3], .. */ + +/* tools/baremetal/hwtcl-vp.s, assembled by tools/isa-usse/usse-as: + * + * 0: mov.skipinv.repeat7 o4, pa4 colour and texcoord through + * 1..4: X, Y, Z, W = m0i*x + m3i + * 5..12: += m*y then m*z + * 13: frcp.skipinv pa7, pa7 + * 14..16: o0..o2 = pa4..pa6 * pa7 the perspective divide + * 17: mov.skipinv o3, c52 c52 is the hardwired 1.0 + * 18: emitvtx.end.freep #0, #0 + */ +static const uint32_t hwtcl_vp[] = { + 0xA0800200u, 0x28A16001u, + 0xA0838011u, 0x00801006u, + 0xA0A48015u, 0x00801006u, + 0xA0C58019u, 0x00801006u, + 0xA0E6801Du, 0x00801006u, + 0xA083C084u, 0x00801006u, + 0xA0A4C085u, 0x00801006u, + 0xA0C5C086u, 0x00801006u, + 0xA0E6C087u, 0x00801006u, + 0xA0840104u, 0x00801006u, + 0xA0A50105u, 0x00801006u, + 0xA0C60106u, 0x00801006u, + 0xA0E70107u, 0x00801006u, + 0x80E00380u, 0x08801002u, + 0x900103B0u, 0x00811005u, + 0x902143B0u, 0x00811005u, + 0x904183B0u, 0x00811005u, + 0x60601A00u, 0x28831001u, + 0xA0200000u, 0xFB275000u, +}; + +/* BSIZE is four bits, so a transfer of more than sixteen dwords has to be + * described as several lines. This is the divisor search the stock driver + * runs at psb_dri.so:0x381c5: the largest burst below sixteen that divides + * the transfer. */ +int sgx_hwtcl_inrec_uniforms(void) +{ + const char *e = HWTCL_ENV("SGX_VTX_INREC"); + + return e && *e && *e != '0'; +} + +/* What the vertex task is known to run. + * + * Not a field and not a documented limit: it is where the measurements sit on + * either side. Programs of a handful of instructions at sixteen temporaries + * render byte-identically to the CPU transform, and glxgears' fixed-function + * lighting program - 87 to 95 instructions, 28 temporaries - engages the path + * and stops the core on its first frame, with no MMU fault and no event, after + * which the driver refuses every submit until the machine is power-cycled. + * Nothing in between has been run, so the budget sits below the one that + * stalls and above what a fixed-function transform with a texture matrix + * needs. A program past it goes to the draw module, which is slow and right. + * + * SGX_HWTCL_MAX_TEMPS and SGX_HWTCL_MAX_INSNS move them, to find where the + * real limit is. Thirty-one is the DOUTU's temporary field and 128 the + * instructions a per-record program copy has room for, so neither is useful + * past that. */ +unsigned sgx_hwtcl_max_temps(void) +{ + static int n = -1; + + if (n < 0) { + const char *e = HWTCL_ENV("SGX_HWTCL_MAX_TEMPS"); + + n = e && *e ? (int)strtol(e, NULL, 0) : 24; + if (n < 1) + n = 1; + if (n > 31) + n = 31; + } + return (unsigned)n; +} + +unsigned sgx_hwtcl_max_insns(void) +{ + static int n = -1; + + if (n < 0) { + const char *e = HWTCL_ENV("SGX_HWTCL_MAX_INSNS"); + int cap = (int)(SGX_HWTCL_PROG_STRIDE / 8u); + + /* The program slot's own capacity, which is the only limit + * here with a reason behind it. The 80 this used to be was a + * guess placed under the 87-to-95 instruction program that + * stalled - but that one also had 28 temporaries, and + * sgx_hwtcl_max_temps() already refuses anything past 24, so + * the case it was guarding against cannot reach here. + * + * Measured over glmark2 with each scene run on its own, since + * the suite in one process contaminates later scenes: gouraud + * goes from 3 fps to 6 and texture:mipmap from 29 to 31, with + * no scene slower and none newly losing frames. glmark2's + * shading vertex program is 86 instructions with 12 + * temporaries, so the old cap refused it by six. */ + n = e && *e ? (int)strtol(e, NULL, 0) : cap; + if (n < 1) + n = 1; + if (n > cap) + n = cap; + } + return (unsigned)n; +} + +int sgx_hwtcl_ntex(void) +{ + const char *e = HWTCL_ENV("SGX_VTX_NTEX"); + unsigned v = e && *e ? (unsigned)strtoul(e, NULL, 0) : 2u; + + return (v >= 1u && v <= XPSB_NTEX_MAX) ? (int)v : 2; +} + +unsigned sgx_hwtcl_record_dwords(unsigned n) +{ + unsigned base = xpsb_vtx_stride((unsigned)sgx_hwtcl_ntex()); + + if (!sgx_hwtcl_inrec_uniforms()) + return base; + if (n > SGX_HWTCL_MAX_UNI) + n = SGX_HWTCL_MAX_UNI; + return base + n; +} + +/* The vertex fetch's DMA control counts dwords and the fetch program + * multiplies the index by the stride in bytes; both come from the record's + * width, which is why they are written together. Past sixteen dwords the DMA + * takes more than one line, the same split sgx_hwtcl_dma_ctl() does. */ +int sgx_hwtcl_set_record(uint32_t *heap, unsigned dwords) +{ + uint32_t ctl; + + if (!heap || dwords < SGX_HWTCL_STRIDE || + dwords > SGX_HWTCL_PA_DEPTH) + return -EINVAL; + ctl = sgx_hwtcl_dma_ctl(dwords, 0); + if (!ctl) + return -EINVAL; + heap[XPSB_VTXDMA_DW] = ctl; + heap[XPSB_VTXSTRIDE_DW] = dwords * 4u; + return 0; +} + +/* Group 10 [28:24]: how many dwords the MTE emits per vertex. On the software + * path this equals the record width, because the vertex program passes the + * record through - which is why xpsb_heap_set_record_width() writes both from + * one number. Under hardware transform they are different numbers: the record + * is what the fetch reads, this is what the vertex program writes. */ +int sgx_hwtcl_set_out_dwords(uint32_t *heap, unsigned dwords) +{ + uint32_t *g10; + + if (!heap || !dwords || dwords > 31u) + return -EINVAL; + g10 = heap + xpsb_heap_state_off(heap, XPSB_STATE_OUTSEL); + *g10 = (*g10 & ~XPSB_STATE_G10_DW) | + ((dwords << XPSB_STATE_G10_SHIFT) & XPSB_STATE_G10_DW); + return 0; +} + +/* On by default. sgx_pipe_settle_hwtcl() decides per program pair whether the + * part can actually run the transform, so this only says the driver may try. + * SGX_HWTCL=0 turns it off outright. */ +int sgx_hwtcl_wanted(void) +{ + const char *e = HWTCL_ENV("SGX_HWTCL"); + + return !(e && *e == '0'); +} + +/* On by default, and hardware TCL needs it: without a copy of the program per + * uniform set, a caller that changes its uniforms inside a frame drops the + * whole context back to the draw module mid-frame, which is what left every X + * client window black. */ +int sgx_hwtcl_limm_uniforms(void) +{ + const char *e = HWTCL_ENV("SGX_VTX_LIMM"); + + return !(e && *e == '0'); +} + +/* LIMM splits its 32 bits across word0[20:0], word1[8:4] and word1[17:12] - + * usse_isa.c's encoder - so patching one is three fields, not a store. */ +unsigned sgx_hwtcl_prog_off(unsigned slot) +{ + return SGX_HWTCL_USSE_OFF + slot * SGX_HWTCL_PROG_STRIDE; +} + +/* The DOUTU's first word is dep | (2 << 4) | ((exe >> 4) << 8) | low, and the + * kernel writes the execution address into bits [18:8] through the relocation + * whose mask is 0x0007ff00. Recomputing that one field is all a copy needs. */ +uint32_t sgx_hwtcl_vpds_copy(uint32_t *heap, uint64_t heap_va, uint64_t use_va, + unsigned slot) +{ + uint32_t off = SGX_HWTCL_VPDS_COPY_OFF + + slot * SGX_HWTCL_VPDS_COPY_STRIDE; + uint32_t *dst = heap + off / 4; + uint64_t exe = use_va + sgx_hwtcl_prog_off(slot); + + memcpy(dst, heap + SGX_HWTCL_VPDS_DATA_OFF / 4, + SGX_HWTCL_VPDS_COPY_DW * 4); + /* ds0[4] is the DOUTU's first word: data dword 4 of the copy. */ + dst[4] = (dst[4] & ~0x0007ff00u) | + (uint32_t)(((exe >> 4) << 8) & 0x0007ff00u); + return (uint32_t)(((heap_va + off) >> 4) & 0x0fffffffu); +} + +/* As sgx_hwtcl_vpds_copy(), but keyed by record and carrying that record's + * vertex width. The DMA control and the byte stride are data dwords 1 and 8, + * which is heap 0x3a4 and 0x3c0 in the frame's own program. */ +uint32_t sgx_vpds_rec_copy(uint32_t *heap, uint64_t heap_va, uint64_t use_va, + unsigned rec, unsigned prog_slot, unsigned nfloat) +{ + uint32_t off = SGX_VPDS_REC_COPY_OFF + + rec * SGX_HWTCL_VPDS_COPY_STRIDE; + uint32_t *dst = heap + off / 4; + uint32_t ctl = xpsb_dma_ctl(nfloat, 0); + + memcpy(dst, heap + SGX_HWTCL_VPDS_DATA_OFF / 4, + SGX_HWTCL_VPDS_COPY_DW * 4); + if (ctl) { + dst[SGX_VPDS_REC_DMACTL_DW] = ctl; + dst[SGX_VPDS_REC_STRIDE_DW] = nfloat * 4u; + } + if (prog_slot != SGX_VPDS_NO_PROG) { + uint64_t exe = use_va + sgx_hwtcl_prog_off(prog_slot); + + dst[4] = (dst[4] & ~0x0007ff00u) | + (uint32_t)(((exe >> 4) << 8) & 0x0007ff00u); + } + return (uint32_t)(((heap_va + off) >> 4) & 0x0fffffffu); +} + +int sgx_hwtcl_patch_uniforms(uint32_t *use, unsigned off, + const struct uir_limm_ref *limm, + unsigned nlimm, const float *v, unsigned n) +{ + unsigned k; + + if (!use || (nlimm && !limm)) + return -EINVAL; + if (HWTCL_ENV("SGX_DEBUG")) { + unsigned q, miss = 0, hi = 0; + + for (q = 0; q < nlimm; q++) { + if (limm[q].dword >= n) + miss++; + if (limm[q].dword > hi) + hi = limm[q].dword; + } + fprintf(stderr, "sgx: limm: %u refs, %u supplied dwords, " + "highest ref %u, %u patched as zero\n", nlimm, n, hi, + miss); + } + for (k = 0; k < nlimm; k++) { + uint32_t *w = use + off / 4 + limm[k].insn * 2; + uint32_t bits = 0; + + if (limm[k].dword < n) + memcpy(&bits, &v[limm[k].dword], 4); + w[0] = (w[0] & ~0x001fffffu) | (bits & 0x001fffffu); + w[1] = (w[1] & ~0x0003f1f0u) | + (((bits >> 21) & 0x1fu) << 4) | + (((bits >> 26) & 0x3fu) << 12); + } + return 0; +} + +int sgx_hwtcl_sa_uniforms(void) +{ + const char *e = HWTCL_ENV("SGX_VTX_SA"); + + return e && *e && *e != '0'; +} + +/* Whether the parameter stream carries the input-state record at all. */ +int sgx_hwtcl_ta_record(void) +{ + return sgx_hwtcl_sa_uniforms() || HWTCL_ENV("SGX_TA_HALT") != NULL; +} + +uint32_t sgx_hwtcl_dma_ctl(unsigned d, unsigned ao) +{ + unsigned bs = d, i; + + if (d > 16) { + for (i = 15; i > 1; i--) + if (d % i == 0 && d / i <= 16) + break; + if (i <= 1) + return 0; + bs = i; + } + return 0x80000000u | ((bs - 1u) << 21) | (((d / bs) - 1u) << 4) | + (bs - 1u) | (ao << 8); +} + +int sgx_hwtcl_install(uint32_t *heap, uint32_t *use, uint64_t heap_va, + unsigned nuni) +{ + uint32_t *code = heap + VPDS_CODE_DW; + uint32_t ctl; + + if (!heap || !use || !nuni) + return -EINVAL; + if (nuni > SGX_HWTCL_MAX_UNI) + return -EINVAL; + ctl = sgx_hwtcl_dma_ctl(nuni, SGX_HWTCL_MAT_AO); + if (!ctl) + return -EINVAL; + + /* Refuse rather than patch blind: these offsets are into a captured + * template, and a template that changed shape would otherwise have + * three of its dwords quietly overwritten. */ + if (code[1] != VPDS_MUL || code[2] != VPDS_MOVSA || + code[3] != XPSB_PDS_HALT) + return -EINVAL; + + memcpy(use + SGX_HWTCL_USSE_OFF / 4, hwtcl_vp, sizeof hwtcl_vp); + /* No uniform DMA when the values do not come from memory: with them + * compiled into the program there is nothing to fetch, and issuing a + * transfer whose source was never written hangs the task. The code + * then stays four dwords, which is what the per-record copies expect. + */ + if (sgx_hwtcl_sa_uniforms() || sgx_hwtcl_limm_uniforms() || + sgx_hwtcl_inrec_uniforms()) + return 0; + heap[VPDS_SRC_DW] = (uint32_t)(heap_va + SGX_HWTCL_MAT_OFF); + heap[VPDS_CTL_DW] = ctl; + code[4] = code[3]; + code[3] = code[2]; + code[2] = PDS_DOUTD_2; + return 0; +} + +void sgx_hwtcl_set_matrix(uint32_t *heap, const float *m, unsigned n) +{ + memcpy(heap + SGX_HWTCL_MAT_OFF / 4, m, n * sizeof *m); +} + +void sgx_hwtcl_default_program(uint32_t *use) +{ + memcpy(use + SGX_HWTCL_USSE_OFF / 4, hwtcl_vp, sizeof hwtcl_vp); +} + +int sgx_hwtcl_set_program(uint32_t *use, uint64_t use_size, + const uint64_t *code, unsigned ninsns) +{ + return sgx_hwtcl_set_program_at(use, use_size, SGX_HWTCL_USSE_OFF, + code, ninsns); +} + +int sgx_hwtcl_set_program_at(uint32_t *use, uint64_t use_size, unsigned off, + const uint64_t *code, unsigned ninsns) +{ + unsigned i; + + if (!use || !code || !ninsns) + return -EINVAL; + /* The hole runs to the end of the object; a program past it would + * overwrite whatever the frame keeps there. */ + if ((uint64_t)off + (uint64_t)ninsns * 8 > use_size) + return -ENOSPC; + if (ninsns * 8 > SGX_HWTCL_PROG_STRIDE) + return -ENOSPC; /* it would run into the next copy */ + for (i = 0; i < ninsns; i++) { + use[off / 4 + i * 2] = (uint32_t)(code[i] & 0xffffffffu); + use[off / 4 + i * 2 + 1] = (uint32_t)(code[i] >> 32); + } + return 0; +} + +int sgx_hwtcl_set_uniforms(uint32_t *heap, uint64_t heap_va, const float *v, + unsigned ndwords) +{ + uint32_t ctl; + + if (!heap || !v || !ndwords) + return -EINVAL; + /* The primary bank's room bounds this: the record occupies pa0..13 and + * the transfer lands at pa14. Lifting it needs the uniforms in the + * secondary bank, which the vendor loads from a separate input-state + * PDS program - not from this per-vertex one, whose second descriptor + * slot is spoken for (work/vertex-pds/README.md section 7). */ + if (ndwords > SGX_HWTCL_MAX_UNI) + return -EINVAL; + /* SGX_VTX_UNI_PAD widens the transfer without changing the shader, + * which is what tells a two-line DMA apart from the bank's depth: a + * mat4 reads pa14..29 either way, and eighteen dwords is two lines and + * still ends inside the bank at pa31. */ + { + const char *e = HWTCL_ENV("SGX_VTX_UNI_PAD"); + unsigned pad = e && *e ? (unsigned)strtoul(e, NULL, 0) : 0; + + if (pad && ndwords + pad <= SGX_HWTCL_MAX_UNI) { + memset(heap + SGX_HWTCL_MAT_OFF / 4 + ndwords, 0, + pad * 4); + ndwords += pad; + } + } + ctl = sgx_hwtcl_dma_ctl(ndwords, SGX_HWTCL_MAT_AO); + if (!ctl) + return -EINVAL; + memcpy(heap + SGX_HWTCL_MAT_OFF / 4, v, ndwords * sizeof *v); + heap[VPDS_SRC_DW] = (uint32_t)(heap_va + SGX_HWTCL_MAT_OFF); + heap[VPDS_CTL_DW] = ctl; + return 0; +} + +/* The vertex PDS program's DOUTU words, which its code names as ds0[4], + * ds0[5] and ds1[1] against a data segment starting at heap +0x3a0. */ +#define VPDS_DOUTU0_DW 0xecu +#define VPDS_DOUTU1_DW 0xedu +#define VPDS_DOUTU2_DW 0xf1u + +int sgx_hwtcl_set_temps(uint32_t *heap, unsigned n) +{ + const char *g = HWTCL_ENV("SGX_VTX_TEMP_GRANT"); + + if (!heap) + return -EINVAL; + /* The emitvtx staging needs registers the program's own count does not + * include, so the grant is raised above it - the field holds 31. */ + if (g && *g) { + unsigned v = (unsigned)strtoul(g, NULL, 0); + + if (v >= n && v <= 31u) + n = v; + } + if (n > 31) + return -ENOSPC; + /* SGX_VTX_DOUTU_VENDOR writes the vendor's hardware-path literals + * instead of a count derived from the program: word 1 = 0xe0000000 + * (a grant of 28, the same value whether its program uses none or + * five) and word 2 = 12, which our software-path heap template leaves + * at zero because it was captured from the software path. A grant + * sized to exactly what the program uses leaves the vertex task no + * headroom for the emitvtx staging. */ + if (HWTCL_ENV("SGX_VTX_DOUTU_VENDOR")) { + heap[SGX_HWTCL_VTX_TEMPS_DW] = + (heap[SGX_HWTCL_VTX_TEMPS_DW] & ~0xf8000000u) | + 0xe0000000u; + { + /* SGX_VTX_DOUTU2 overrides word 2. pds_isa.c:922 reads + * its [3:0] as the temporary count, which would make + * the vendor's 12 a grant - and a program needing + * thirteen would fail on exactly that. */ + const char *e = HWTCL_ENV("SGX_VTX_DOUTU2"); + + heap[VPDS_DOUTU2_DW] = e && *e ? + (uint32_t)strtoul(e, NULL, 0) : + 0x0000000cu; + } + return 0; + } + heap[SGX_HWTCL_VTX_TEMPS_DW] = + (heap[SGX_HWTCL_VTX_TEMPS_DW] & ~0xf8000000u) | (n << 27); + return 0; +} + +/* The vertex PDS program's DOUTU words, which its code names as ds0[4], + * ds0[5] and ds1[1] against a data segment starting at heap +0x3a0. */ + +#define PDS_DOUTD_0 0x07030223u /* movs doutd, ds0[0], ds0[1], .. */ +#define PDS_DOUTU_2 0x07042345u /* movs doutu, ds0[2], ds0[3], ds1[2] */ + +int sgx_hwtcl_input_const_state(uint32_t *heap, uint64_t heap_va, + uint32_t *use, uint64_t use_va, + uint32_t *rec, const float *v, unsigned n) +{ + uint32_t *prog = heap + SGX_HWTCL_SEC_PDS_OFF / 4; + uint32_t ctl; + + if (!heap || !rec || !v || !n) + return -EINVAL; + if (n > 16) + return -ENOSPC; /* one burst; the split is not built */ + /* SGX_TA_PA_UNI aims this program's DMA at the primary bank, where the + * frame's own uniform DOUTD puts them and where the shader is still + * compiled to read them. If the picture comes out right, the program + * runs and its DMA lands in pa - which would mean this route cannot + * reach the secondary bank at all. */ + ctl = sgx_hwtcl_dma_ctl(n, HWTCL_ENV("SGX_TA_PA_UNI") ? + SGX_HWTCL_MAT_AO : 0u); + if (!ctl) + return -EINVAL; + + /* The control: emit the record but bind a program that does nothing, + * with the uniforms left where they work. If the frame still renders, + * the record, the stream prefix and the relocation shift are all + * sound and only the program is at fault. */ + /* SGX_TA_NOREC keeps the two-dword reservation and the relocation + * shift but writes no record into it, which separates a stream the + * parser cannot walk from a record it cannot read. */ + if (HWTCL_ENV("SGX_TA_NOREC")) { + rec[0] = 0; + rec[1] = 0; + return 0; + } + if (HWTCL_ENV("SGX_TA_HALT")) { + memset(prog, 0, (SGX_HWTCL_SEC_PDS_DATA + 4) * 4); + prog[SGX_HWTCL_SEC_PDS_DATA] = XPSB_PDS_HALT; + rec[0] = 0x40000000u | + (uint32_t)(((heap_va + SGX_HWTCL_SEC_PDS_OFF) >> 4) & + 0x0fffffffu); + rec[1] = ((SGX_HWTCL_SEC_PDS_DATA << 24) & 0xfc000000u) | + SGX_HWTCL_INPUT_STATE | 1u; + return 0; + } + + memcpy(heap + SGX_HWTCL_MAT_OFF / 4, v, n * sizeof *v); + memset(prog, 0, (SGX_HWTCL_SEC_PDS_DATA + 4) * 4); + prog[0] = (uint32_t)(heap_va + SGX_HWTCL_MAT_OFF); + prog[1] = ctl; + /* A nop for this DOUTU to launch. The vertex shader is launched by the + * per-vertex program; launching it here as well ran the whole + * transform once with nothing fetched into the primary bank, and the + * emitvtx at its end put a garbage vertex into the MTE outside any + * primitive block. The vendor's program launches a one-instruction + * nop.end here for the same reason. */ + if (use) { + use[SGX_HWTCL_NOP_OFF / 4] = 0x00000000u; + use[SGX_HWTCL_NOP_OFF / 4 + 1] = 0xf8040140u; + } + prog[2] = (heap[VPDS_DOUTU0_DW] & ~0x0007ff00u) | + (uint32_t)(((((use_va + SGX_HWTCL_NOP_OFF) >> 4) << 8)) & + 0x0007ff00u); + /* Bit 26 is set on the vendor's path whose shader reads the secondary + * bank; the rest of the word is its own, not the vertex program's. */ + prog[3] = 0x04000000u; + prog[10] = 0u; + prog[SGX_HWTCL_SEC_PDS_DATA + 0] = PDS_DOUTD_0; + prog[SGX_HWTCL_SEC_PDS_DATA + 1] = PDS_DOUTU_2; + prog[SGX_HWTCL_SEC_PDS_DATA + 2] = XPSB_PDS_HALT; + + rec[0] = 0x40000000u | + (uint32_t)(((heap_va + SGX_HWTCL_SEC_PDS_OFF) >> 4) & + 0x0fffffffu); + /* The low byte is the size of the block the program loads, in 16-byte + * units - the secondary attributes the vertex shader will read, not + * the program's own code. Hardcoding it to one declared four + * registers while the DOUTD writes n dwords and the shader reads + * sa0..sa[n-1], so the shader named registers the task had no room + * for and the tiler never finished. */ + rec[1] = ((SGX_HWTCL_SEC_PDS_DATA << 24) & 0xfc000000u) | + SGX_HWTCL_INPUT_STATE | (((n * 4u + 15u) >> 4) & 0xffu); + return 0; +} diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_hwtcl.h mesa-26.2.2/src/gallium/drivers/sgx/sgx_hwtcl.h --- mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_hwtcl.h 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/sgx_hwtcl.h 2026-09-08 10:57:36.680183148 +0200 @@ -0,0 +1,252 @@ +/* Hardware vertex transform: the USSE runs the vertex program instead of + * Mesa's draw module. + * + * Copyright (C) 2026 RenĂ© Rebe + * SPDX-License-Identifier: MIT + */ +#ifndef SGX_HWTCL_H +#define SGX_HWTCL_H + +#include + +#include "usse_ir.h" + +/* Where the vertex program and its uniforms live. Both are holes in the + * captured heap and USSE templates that no relocation touches, which is why + * they can be written after generation. Same values as + * tools/baremetal/sgxtri.c, so a frame from either is the same frame. */ +/* Above the captured USSE template (which ends near 0x2060) and above the + * region a long fragment program uses, which ends at 0x8000. This sat at + * 0x280, inside the template, where a program longer than sixteen + * instructions overwrote the captured objects behind it. */ +#define SGX_HWTCL_USSE_OFF 0x8000u +#define SGX_HWTCL_MAT_OFF 0x0480u +#define SGX_HWTCL_MAT_N 16u +#define SGX_HWTCL_MAT_AO 14u /* just past the 14-dword record */ +/* The viewport the epilogue reads, in the dwords the attributes leave free: + * scale x, translate x, scale y, translate y, scale z, translate z. */ +#define SGX_HWTCL_VP_AO 8u +/* The primary attribute bank is 32 registers deep and the record occupies the + * first fourteen, so this is every uniform dword a vertex program can have. + * The stock driver's hardware path instead has a second PDS program put its + * constants in the secondary bank, which would lift this - see + * work/vertex-pds/README.md section 7. Not established on hardware here. */ +/* How deep the primary attribute bank is, measured rather than assumed: a + * vertex program whose uniforms end at pa29 renders correctly and one whose + * uniforms end at pa33 does not, whatever heap dword 0xf1 - the field a + * capture shows changing on the closed driver's hardware path - is set to. + * The register field in the instruction encoding is seven bits, so the limit + * is what the task is given, and nothing found so far raises it. */ +#define SGX_HWTCL_PA_DEPTH 32u +#define SGX_HWTCL_MAX_UNI (SGX_HWTCL_PA_DEPTH - SGX_HWTCL_MAT_AO) + +/* The vertex record hardware TCL needs: position 4, colour 4, texcoord 3 and + * the second unit's 3, which is where a normal would ride. */ +#define SGX_HWTCL_STRIDE 14u + +/* One DOUTD control word for d dwords landing at attribute offset ao. */ +uint32_t sgx_hwtcl_dma_ctl(unsigned d, unsigned ao); + +/* Install the vertex program and grow the vertex PDS program by the DOUTD + * that loads the uniforms. heap_va is where the heap object is mapped on the + * GPU. Returns 0, or -EINVAL if the PDS program is not the shape this patch + * expects - which means the template changed and the offsets below are stale. + */ +int sgx_hwtcl_install(uint32_t *heap, uint32_t *use, uint64_t heap_va, + unsigned nuni); + +/* This frame's transform, row major: row i dotted with (x, y, z, 1) is clip + * component i, and row 3 the divisor. */ +void sgx_hwtcl_set_matrix(uint32_t *heap, const float *m, unsigned n); + +/* The vertex task's temporary-register count, bits [31:27] of the vertex PDS + * program's DOUTU payload at heap dword 0xed - the vertex-side twin of the + * fragment task's XPSB_HEAP_USE_TEMPS. It is zero in the captured frame + * because the pass-through program uses no temporaries, and a compiled + * program that writes r0 upwards then writes out of range. + * work/tcl-capture/ sees the closed driver set the same field to 0xe0000000, + * i.e. 28, on its hardware-transform path. */ +#define SGX_HWTCL_VTX_TEMPS_DW 0xedu + +/* Put a compiled vertex program where the retargeted USE records point. Two + * dwords an instruction. Returns -ENOSPC if it would not fit the hole. */ +int sgx_hwtcl_set_program(uint32_t *use, uint64_t use_size, + const uint64_t *code, unsigned ninsns); + +/* The same at a chosen offset, for the per-record copies. */ +int sgx_hwtcl_set_program_at(uint32_t *use, uint64_t use_size, unsigned off, + const uint64_t *code, unsigned ninsns); + +/* How many temporaries the vertex task may use. */ +int sgx_hwtcl_set_temps(uint32_t *heap, unsigned n); + +/* Uniforms in the secondary attribute bank instead of the primary one, which + * is what lifts the eighteen-dword ceiling the primary bank imposes. + * SGX_VTX_SA turns it on. */ +int sgx_hwtcl_sa_uniforms(void); + +/* Uniforms materialised as LIMM immediates in the vertex program instead of + * read from an attribute bank. The primary bank is 32 registers with the + * vertex record in front of it, which caps a program at eighteen uniform + * dwords, and the secondary bank is not reachable on this stage - an + * immediate needs no register at all. The driver rewrites the program every + * flush anyway, so the values are patched in rather than recompiled. + * SGX_VTX_LIMM. */ +int sgx_hwtcl_limm_uniforms(void); + +/* Whether the part is to run the vertex program. Constant for a context. */ +int sgx_hwtcl_wanted(void); + +/* The largest vertex program the part is known to run: temporaries and + * instructions. A program past either is refused at compile time and runs on + * the draw module instead, because a vertex task that cannot run does not + * fail - it stops the core, and the machine has to be power-cycled. + * SGX_HWTCL_MAX_TEMPS and SGX_HWTCL_MAX_INSNS override them. */ +unsigned sgx_hwtcl_max_temps(void); +unsigned sgx_hwtcl_max_insns(void); + +/* Uniforms carried in the vertex record, one copy per vertex, at the dwords + * just past the record's own fourteen - which is where the uniform DMA used to + * put them, so the compiled program and the epilogue are unchanged. + * + * This is what gives a frame a matrix per object. A record's uniforms are + * fixed for the whole record, so per-object values otherwise need a draw + * record each, and a frame with two of those stalls on its third frame - see + * driver/mesa/README.md. Per vertex, one record carries them all. + * + * The primary attribute bank is 32 registers and the record takes the first + * fourteen, so this holds the same eighteen dwords the DMA did. SGX_VTX_INREC. + */ +int sgx_hwtcl_inrec_uniforms(void); +/* Coordinate sets the hardware path's record carries, and so its width. + * SGX_VTX_NTEX overrides it: the built-in transform reads its viewport from + * the uniforms rather than the record, so it runs on the narrow shape too. */ +int sgx_hwtcl_ntex(void); + +/* The record's width in dwords when it carries n uniform dwords. */ +unsigned sgx_hwtcl_record_dwords(unsigned n); + +/* Point the vertex fetch at a record of that width. */ +int sgx_hwtcl_set_record(uint32_t *heap, unsigned dwords); +/* How many dwords the MTE emits per vertex, into group 10. */ +int sgx_hwtcl_set_out_dwords(uint32_t *heap, unsigned dwords); + +/* How many uniform dwords a program can carry in its code. */ +#define SGX_HWTCL_MAX_UNI_LIMM UIR_MAX_LIMM + +/* Distinct uniform sets one frame may hold. A draw whose uniforms differ from + * the open record's starts a new record, and each record's vertex program is + * its own copy with its own values compiled in - which is what lets a frame + * carry several, as a scene with a matrix per object does. + * + * Thirty-one is what the USSE object's layout allows, not a chosen number: + * the copies start at SGX_HWTCL_USSE_OFF and are SGX_HWTCL_PROG_STRIDE apart, + * and the per-record fragment programs start at 0x10000, so the last copy + * that fits below them is the thirty-first. Going past it is a refused draw, and a + * refused draw drops the whole frame. */ +#define SGX_HWTCL_MAX_UNI_SETS 31u + +/* Bytes between the per-record vertex program copies in the USSE object, and + * where they start. The first copy is the program itself. */ +/* Doubled from 0x400. That held 128 instructions and was what refused + * glmark2's refract vertex program, which is 129 - so the scene ran its + * transform in the draw module's interpreter and spent 5.6 seconds a frame at + * 87% of a core, the same at 320x240 as at 1280x720. The copies run from + * SGX_HWTCL_USSE_OFF and must stay below the per-record fragment programs at + * SGX_USSE_GROUP_OFF, which moved up to make the room: 0x8000 + 31 * 0x800 is + * 0x17800, under 0x20000. */ +#define SGX_HWTCL_PROG_STRIDE 0x800u + +/* Where the per-record vertex PDS program copies go. Each is the frame's own + * program - twelve data dwords at heap +0x3a0 and its code - copied verbatim + * so its vertex fetch stays intact, with only the DOUTU's execution address + * moved to that record's program copy. */ +/* The frame's own vertex PDS program: data segment and its dword count. */ +#define SGX_HWTCL_VPDS_DATA_OFF 0x03a0u +/* Well above the captured heap, alongside the other per-record copy regions. + * This sat at 0x600, which is where a fragment program's constant block sits + * (SGX_FS_CONST_OFF) - a frame with hardware transform and a fragment shader + * that has constants wrote both into the same bytes. Above every per-record + * region; sgx_context.c checks that it still is. */ +#define SGX_HWTCL_VPDS_COPY_OFF 0x6a0000u +/* Twenty, not sixteen: every PDS program in the heap template is followed by + * a bare HALT four dwords past its code - heap 0xf4 is the vertex program and + * 0xf8 the HALT behind it - and a copy that stops before it does not run. So + * a copy is the twelve data dwords, the four code dwords and that HALT. */ +#define SGX_HWTCL_VPDS_COPY_DW 20u +#define SGX_HWTCL_VPDS_COPY_STRIDE (32u * 4u) + +/* A copy per draw record rather than per uniform set. The vertex fetch's DMA + * control and byte stride are data dwords 1 and 8 of this program, so a record + * with its own copy names its own record width - which is what let the frame + * stop ending every time a draw asked for a different one. Keyed by record + * because a record needs both its width and, under hardware transform, its + * uniform set's execution address, and those do not share a key. */ +#define SGX_VPDS_REC_COPY_OFF 0x6b0000u +#define SGX_VPDS_REC_DMACTL_DW 1u +#define SGX_VPDS_REC_STRIDE_DW 8u + +/* Copy the vertex PDS program for record `rec`, giving it `nfloat` floats of + * record and, when `prog_slot` is not SGX_VPDS_NO_PROG, aiming its DOUTU at + * that uniform set's program copy. Returns the record's word 5. */ +#define SGX_VPDS_NO_PROG ((unsigned)-1) +uint32_t sgx_vpds_rec_copy(uint32_t *heap, uint64_t heap_va, uint64_t use_va, + unsigned rec, unsigned prog_slot, unsigned nfloat); + +/* Copy the vertex PDS program for record `slot`, aiming its DOUTU at that + * record's program copy, and return the draw-record word 5 that binds it. */ +uint32_t sgx_hwtcl_vpds_copy(uint32_t *heap, uint64_t heap_va, + uint64_t use_va, unsigned slot); + +/* Where record `slot`'s vertex program copy sits in the USSE object. */ +unsigned sgx_hwtcl_prog_off(unsigned slot); + +/* Patch the materialised uniforms of a program already written to the USSE. */ +int sgx_hwtcl_patch_uniforms(uint32_t *use, unsigned off, + const struct uir_limm_ref *limm, + unsigned nlimm, const float *v, unsigned n); +int sgx_hwtcl_ta_record(void); + +/* The parameter stream is a sequence of self-describing records and word 1's + * constant says which kind: 0x0002a200 opens a draw block, 0x0001e100 is the + * input-state record psb_ta_input_state (psb_dri.so 0x3b606) emits. Its only + * caller is the tail of emit_input_const_state, the function that builds the + * program which loads a vertex shader's constants - and that program ends in + * a DOUTU, so it launches the shader itself rather than only feeding it. Two + * dwords, ahead of every draw record. */ +#define SGX_HWTCL_TA_PREFIX_DW 2u +#define SGX_HWTCL_INPUT_STATE 0x0001e100u + +/* Where that program is built. The heap template writes nothing above dword + * 0xfb, the fragment task's moved null program takes 0x400 and the uniforms + * 0x480, so this block is free. Its data segment is twelve dwords - the size + * has to be a multiple of four, psb_ta.c:287 - and ds0[k] is data dword k + * while ds1[k] is data dword 8 + k, the split the capture's own block + * resolves. */ +#define SGX_HWTCL_SEC_PDS_OFF 0x0500u +#define SGX_HWTCL_SEC_PDS_DATA 12u + +/* Build it: a DOUTD that loads n uniform dwords into the secondary bank at + * attribute offset 0, then the DOUTU that launches the vertex program, taken + * from the three words this frame's own vertex PDS program uses - heap dwords + * 0xec, 0xed and 0xf1 - so it launches exactly what the draw record's program + * would have. Writes the two-dword input-state record into rec[0..1]. */ +int sgx_hwtcl_input_const_state(uint32_t *heap, uint64_t heap_va, + uint32_t *use, uint64_t use_va, + uint32_t *rec, const float *v, unsigned n); + +/* A one-instruction `nop.end` for the input-state program's DOUTU to launch. + * That program feeds the secondary bank; the vertex shader is launched by the + * per-vertex program, so launching it here too runs the whole transform once + * with no vertex fetched. Above the captured template, below the region a long + * fragment program uses. */ +#define SGX_HWTCL_NOP_OFF 0x3000u + +/* How many uniform dwords the DOUTD carries, and where they come from. */ +int sgx_hwtcl_set_uniforms(uint32_t *heap, uint64_t heap_va, + const float *v, unsigned ndwords); + +/* The built-in transform-only program, for a caller that has none. */ +void sgx_hwtcl_default_program(uint32_t *use); + +#endif diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_pipe.c mesa-26.2.2/src/gallium/drivers/sgx/sgx_pipe.c --- mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_pipe.c 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/sgx_pipe.c 2026-09-08 10:57:36.680249781 +0200 @@ -0,0 +1,9556 @@ +/* The Gallium entry points: pipe_screen and pipe_context. + * + * Everything this calls into was written and tested without Mesa - the state + * translation, the screen's limits, shader compilation, sampler views, + * resources, and the context that binds them. This file is the adapter: it + * fills in Mesa's vtables and converts between Mesa's enumerations and the + * hardware's. + * + * It is deliberately thin. A bug here is a wrong assignment; a bug in what it + * calls is a wrong frame. Keeping the two apart is why the layer below does + * not include a single Mesa header, and why it can be tested on a machine with + * no Mesa tree at all. + * + * This file does need Mesa, so it is built only when MESA_INCLUDE points at a + * tree. See the note in the Makefile. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#include "pipe/p_screen.h" +#include "pipe/p_context.h" +#include "pipe/p_state.h" +#include "pipe/p_defines.h" +#include "util/format/u_formats.h" +#include "util/format/u_format.h" +#include "util/u_screen.h" +#include "util/u_memory.h" +#include "util/u_inlines.h" +#include "frontend/winsys_handle.h" + +#include +#include +#include +#include "util/u_math.h" +#include "util/u_prim.h" +#include "draw/draw_context.h" +#include "nir/nir_to_tgsi.h" +#include "compiler/nir/nir.h" +#include "tgsi/tgsi_dump.h" +#include "util/u_upload_mgr.h" +#include "util/u_threaded_context.h" +#include "util/slab.h" +#include "util/u_transfer.h" +#include "util/u_framebuffer.h" +#include "util/u_helpers.h" +#include "draw/draw_vbuf.h" + +#include "sgx_pipe_vbuf.h" +#include "util/u_blitter.h" +#include "util/u_surface.h" +#include "util/u_dump.h" +#include "sgx_hwtcl.h" +#include "sgx_twod.h" +#include "sgx_drm.h" + +#include "sgx_context.h" +#include "sgx_screen.h" +#include "sgx_resource.h" +#include "xpsb_frame.h" +#include "sgx_sampler.h" +#include "tgsi_parse.h" + +#include +#include +#include +#include + +/* getenv() is a linear scan of the environment and glibc's does a strncmp per + * entry; these knobs sit in per-draw paths, and asking per draw measured at + * nearly two fifths of ioquake3's CPU. The answer cannot change during a run, + * so each site remembers it. */ +#define SGX_ENVS(name) __extension__({ \ + static const char *sgx_envs_cached_; \ + static int sgx_envs_got_; \ + if (!sgx_envs_got_) { \ + sgx_envs_cached_ = getenv(name); \ + sgx_envs_got_ = 1; \ + } \ + sgx_envs_cached_; \ +}) + + +struct sgx_transfer { + /* threaded_transfer for the same reason the resource is a + * threaded_resource: pipe_transfer is its first member. */ + struct threaded_transfer b; + unsigned face; + void *level_base; + uint32_t level_stride, level_rows; +}; + +struct sgx_pipe_screen { + struct pipe_screen base; + int fd; + struct sgx_winsys *ws; + /* The threaded context allocates every transfer from this, so it is + * sized for the driver's own transfer and not for pipe_transfer. */ + struct slab_parent_pool transfer_pool; + /* Its contexts, so a resource being destroyed can be taken out of + * whatever still names it. Single-threaded, as this driver is. */ + struct sgx_pipe_context *ctxs; + /* Which of the ISP's eight visibility counters are spoken for. The + * kernel's running totals are per open file and the screen is what + * holds the file, so the pool has to be here rather than per context: + * two contexts on one screen counting into the same register would + * each read the other's draws. */ + unsigned vis_used; +}; + +struct sgx_pipe_velems; + +/* Where one vertex element's data is, for the length of a draw. None of it + * can change within one, so the fetch below resolves it once rather than for + * every corner of every triangle. */ +struct sgx_hwtcl_elem { + const char *base; + size_t off; /* to the element's first vertex */ + unsigned stride; + unsigned ncomp; + unsigned char half; /* components are half floats */ + unsigned char u8n; /* components are normalised bytes */ + unsigned char bgra; /* ... and in B, G, R, A order */ + size_t limit; /* 0 when the buffer carries no size */ +}; + +struct sgx_pipe_context { + /* First: every pipe_context pointer the state tracker hands back is + * cast straight to this. */ + struct pipe_context base; + struct sgx_context ctx; + struct sgx_pipe_screen *screen; + struct sgx_pipe_context *next; + /* The loss generation this context has already reported. */ + unsigned reset_gen; + + /* The vertex stage: the caller's vertex shader runs here, and what + * comes out is what the hardware's pass-through program expects. */ + struct draw_context *draw; + struct vbuf_render *vbuf; + /* The staging record sgx_hwtcl_draw() builds, kept across draws. + * It was calloc'd and freed per draw - 1.2 MB for a seven thousand + * triangle model - so every draw took fresh pages from the kernel and + * paid the fault and the kernel's zeroing for them. */ + float *hwrec; + size_t hwrec_floats; + /* An indexed transform draw's index array, and the most corners one + * may name: past that the record block is chunked and an indexed + * chunk would still reach vertices outside it, so the per-corner form + * is taken instead. */ + uint16_t *hwidx; + unsigned hwidx_n; + unsigned hwidx_max_n; + /* The de-duplication's two halves: the open-addressed table, and the + * caller's vertex behind each record it produced. */ + unsigned *hwmap, *hwsrc; + unsigned hwmap_n, hwsrc_n; + /* Each vertex element of the current draw, resolved once. Grown with + * the element count rather than fixed, so a draw is never silently + * truncated to a compile-time maximum. */ + struct sgx_hwtcl_elem *hwel; + unsigned hwel_n; + + struct pipe_framebuffer_state fb; + /* Samples a pixel in the bound framebuffer, 1 or 4, and the last mask + * the state tracker set. sgx_sample_mask_blocks() is the decision taken + * from the two - see sgx_pipe_set_sample_mask(). */ + unsigned fb_samples; + unsigned sample_mask; + struct pipe_viewport_state viewport; + /* Which sampler units carry an sRGB view. The conversion GL requires + * on such a sample is compiled into the fragment program, so a change + * here means the program is built again. */ + unsigned srgb_mask; + /* And which carry a depth view, whose texel the program reassembles. */ + unsigned depth_mask; + /* And what each unit's view makes of the channels it fetched, for the + * same reason - grown to the highest unit the context has bound. */ + uint16_t *swz; + unsigned nswz; + /* And what each unit's fetch returns (SGX_TEXKEY), and the shadow + * comparison bound on it (SGX_CMPKEY) - both baked into the + * program as well, the comparison ahead of the TGSI. */ + unsigned char *cls; + unsigned ncls; + uint32_t *cmp; + unsigned ncmp; + void *vs; /* struct sgx_pipe_vs * */ + int hwtcl_ok; /* both programs suit the part */ + + struct pipe_vertex_buffer vb[PIPE_MAX_ATTRIBS]; + unsigned nvb; + struct sgx_pipe_velems *velems; + + /* Hardware transform. Off unless the caller's vertex shader compiled + * and its uniforms fit the primary attribute bank; see sgx_hwtcl.h. */ + int hwtcl; + /* Frames that opened on the part and then met a draw it could not + * take, so the frame had to be sent early. A caller whose frames keep + * doing that pays a submit for a transform it barely uses, and is + * better off staying on the draw module. */ + unsigned hwtcl_fallbacks; + int hwtcl_unprofitable; + /* What the caller asked for, kept apart from `hwtcl` above: a draw the + * part cannot take turns the transform off for the rest of the frame, + * and this is what turns it back on for the next one. */ + int hwtcl_want, hwtcl_rearm; + unsigned split_every, draws_seen; + float vs_const[SGX_HWTCL_MAX_UNI_LIMM]; + unsigned vs_const_n; + /* The caller's buffer was longer than vs_const_n; the hardware path + * would render with values it never received. */ + int vs_const_trunc; + + /* What a frame can hold one of. A frame is a whole render with one + * fragment program, one texture and one blend, so a draw that needs a + * different one of any of them starts a new frame; these are what the + * pending frame was begun with. */ + void *cur_fs, *cur_blend, *frame_fs, *frame_blend; + /* The ISP words are the record's own, so a change of depth/stencil + * state, rasterizer state or stencil reference opens a new one. */ + void *frame_dsa, *frame_rast; + struct pipe_stencil_ref frame_sref; + /* Bumped whenever the fragment constant buffer is set, so a draw can + * tell that the uniforms it will render with are not the ones the + * open record was built for. */ + unsigned fs_const_serial, frame_fs_const_serial; + /* glBlendColor, packed, and the same for it: the constant is in the + * blend's instructions, so a record carries one. */ + uint32_t blend_const; + unsigned blend_const_serial, frame_blend_const_serial; + struct pipe_resource *cur_tex, *frame_tex; + + /* A copy between two surfaces is a draw on this part, not a memcpy - + * the CPU cannot read the write-combining mapping at any speed - so + * the blitter runs one. It binds its own state and expects the driver + * to put the caller's back, which means the driver has to know what + * the caller had: these are the Gallium objects as bound, kept for no + * other purpose. */ + struct blitter_context *blitter; + void *cur_dsa, *cur_rast; + void *cur_samplers[SGX_MAX_TEX_UNITS]; + unsigned ncur_samplers; + /* A border-mode sampler bound to a unit whose texture has no border + * map is bound to the context as clamp-to-edge instead; this is which + * units hold that substitute, so a texture that does have one gets + * the real state back. */ + unsigned char border_sub[SGX_MAX_TEX_UNITS]; + struct pipe_sampler_view *cur_views[SGX_MAX_TEX_UNITS]; + /* The buffer address each unit's descriptor was built from. A + * resource that is displaced from the depth or target slot moves, and + * the descriptor still names where it used to be. */ + uint64_t view_base[SGX_MAX_TEX_UNITS]; + /* The vertex stage's own, which the draw module samples on the CPU: + * the part has no vertex texture unit, so a vertex fetch is the + * module's to do. Kept apart from the fragment stage's above. */ + struct pipe_sampler_view *vtx_views[SGX_MAX_VTX_SAMPLERS]; + void *vtx_samplers[SGX_MAX_VTX_SAMPLERS]; + unsigned nvtx_views; + unsigned nvtx_samplers; + unsigned ncur_views; + struct pipe_scissor_state scissor_state; + struct pipe_stencil_ref stencil_ref; + /* The fragment constants as the caller set them. The blitter binds its + * own and restores from here; without it every draw after a copy read + * the blitter's uniforms, which is why shaders that touch no texture + * at all came back the wrong colour. */ + struct pipe_constant_buffer cur_fs_cb; + int in_blit; /* no recursion back into the blitter */ + /* The sampler unit the draw module's stipple stage puts its pattern + * on, from the last variant it had built, or -1. The stage restores + * the caller's sampler state by count and so never unbinds the unit; + * sgx_pipe_draw_vbo() does, from what was bound before the draw. */ + int stipple_unit; + + /* The occlusion query whose draws are being counted, and whether the + * state tracker has paused counting (::set_active_query_state, which a + * blit and the mipmap generator turn off around their own draws). One + * at a time: ISP word B names a single counter per object, so two + * queries cannot both have the same draw. */ + struct sgx_pipe_query *vis_active; + int query_paused; +}; + +/* One occlusion query. + * + * The counter it is given is a hardware register index, held from + * ::begin_query until the result has been collected - the count is the + * difference between the kernel's running total then and now, so nothing else + * may use the register in between. Eight of them, so eight queries may be in + * flight; see sgx_pipe_begin_query() for what a ninth is told. */ +struct sgx_pipe_query { + /* First, and zeroed: the threaded context keeps its own state for + * every query and reads it from the front of the driver's. */ + struct threaded_query tq; + unsigned type; /* PIPE_QUERY_OCCLUSION_* */ + int reg; /* -1 until begin_query takes one */ + uint64_t base; /* the running total when it began */ + uint64_t result; + int ended, have_result; +}; + +/* Which transfer path a workload takes. Counted rather than traced: the + * interesting number is how many times per operation, and a line per call + * buries it. SGX_XFER_STATS prints them at every flush. */ +static unsigned sgx_xfer_texmap, sgx_xfer_bufmap, sgx_xfer_texsub; +static unsigned sgx_xfer_bufsub, sgx_xfer_copy, sgx_xfer_flushres; + +/* Why a copy did not reach the engine. Silence is not an answer: a run where + * the 2D path is enabled and never once used looks exactly like a run where + * it is used and gets the pixels wrong, and telling those apart took a + * hardware round trip that this makes unnecessary. */ +enum sgx_twod_why { + SGX_TWOD_TOOK, SGX_TWOD_OFF, SGX_TWOD_FORMAT, SGX_TWOD_SURF, + SGX_TWOD_BOUNDS, SGX_TWOD_SMALL, SGX_TWOD_REFUSED, SGX_TWOD_WHY_N +}; +static const char * const sgx_twod_why_name[SGX_TWOD_WHY_N] = { + "took", "off", "format", "surface", "bounds", "small", "refused" +}; +static unsigned sgx_twod_why[SGX_TWOD_WHY_N]; + +static int sgx_twod_no(enum sgx_twod_why why) +{ + sgx_twod_why[why]++; + return 0; +} + +/* A resource is a Gallium object in front of one of ours. The layout - and in + * particular the stride, which the hardware derives from the width rather than + * being told - is entirely the layer below's business. */ +struct sgx_pipe_resource { + /* threaded_resource, not pipe_resource: the threaded context requires + * a buffer to carry its own bookkeeping, and pipe_resource is its + * first member so every cast through this stays what it was. */ + struct threaded_resource b; + struct sgx_resource r; + struct sgx_winsys *ws; + /* The caller's format has no alpha while the one it is stored in + * does - X8R8G8B8 kept in an A8R8G8B8 surface, because the part has + * no format code without alpha. GL says such a texture samples alpha + * as one, and the byte behind it is whatever the upload left, so it + * is filled in on the way past. */ + unsigned no_alpha; + /* Write mappings open on this buffer. Mesa's upload manager takes one + * and keeps it, writing through it for draw after draw without ever + * unmapping, so a copy taken once is stale from the next write on. */ + unsigned write_mapped; + /* A depth resource's own format, for the CPU's view of it. The + * surface holds what the ISP stores - SGX_DEPTH_SURF_FMT - whatever + * the caller named, and a map converts through this staging copy: + * OES_depth_texture data arrives as Z24 or Z16 and glReadPixels wants + * it back that way. zrow is the row of floats the two meet in. */ + enum pipe_format zfmt; + void *zstage; + unsigned zstage_size; + float *zrow; + unsigned zrow_n; + struct pipe_box zbox; + unsigned zlevel; +}; + +/* The kernel's submit does not return until the render has ended, so by the + * time anything can hold a fence the work behind it is done. There is one + * fence object, shared and never signalled late, rather than a per-submit one + * that would only ever be asked a question it already knows the answer to. */ +/* SGX_DEBUG=1 reports what the adapter did with each call it could not + * otherwise report on: a shader that did not compile and a draw that was + * refused are both silent from above, and both show as an empty frame. */ +static int sgx_aniso_enabled(void); +static int sgx_linear_mips(void); + +static int sgx_debug(void) +{ + static int on = -1; + + if (on < 0) { + const char *e = SGX_ENVS("SGX_DEBUG"); + + on = e && *e && *e != '0'; + } + return on; +} + +/* SGX_HUD_TRACE=1 reports the draws that make up a 2D overlay: the ones the + * draw module hands over inside a band of the window, and every draw the + * hardware path could not place. */ +static int sgx_hud_trace(void) +{ + static int on = -1; + + if (on < 0) { + const char *e = SGX_ENVS("SGX_HUD_TRACE"); + + on = e && *e && *e != '0'; + } + return on; +} + +/* Arguments are evaluated only when debugging is on, so nothing that has to + * happen may be written as one. That is not a style note: a call passed to + * this as an argument stops happening the moment SGX_DEBUG is unset, and this + * driver has lost a framebuffer bind and a sampler view to exactly that, each + * time appearing as a feature that worked under debugging and not otherwise. */ +#define sgx_dbg(...) do { if (sgx_debug()) fprintf(sgx_log(), "sgx: " __VA_ARGS__); } while (0) + +/* Whether a switch that is on by default was turned off. */ +static int sgx_pipe_env_off(const char *name) +{ + const char *e = getenv(name); + + return e && *e == '0'; +} + +struct sgx_pipe_rast { + struct pipe_rasterizer_state pipe; + struct sgx_rasterizer_state hw; + int scissor; /* pipe_rasterizer_state::scissor */ + /* The sprite state the draw module is not shown: the vbuf widens + * points and writes the coordinates itself. */ + unsigned sprite_enable; + int sprite_upper_left; +}; + +/* Whether a program reading gl_FrontFacing has to invert the ISP's bit: the + * bit is set for a clockwise device-space triangle, and which winding is front + * there is front_ccw XOR the target's flip - the same three booleans the cull + * word is built from (sgx_raster_word()). */ +static unsigned sgx_face_swap_of(const struct sgx_rasterizer_state *r) +{ + /* Measured: the unflipped sense drew ccw red and cw green under + * both winding conventions, so the part's bit reads the other way. */ + return (r->front_ccw ^ r->target_y_flipped) & 1u; +} + +struct sgx_pipe_velems { + struct pipe_vertex_element el[PIPE_MAX_ATTRIBS]; + unsigned n; +}; + +/* A vertex shader is two things at once: the draw module's, which runs it on + * the CPU, and a compiled USSE program, which runs it on the part. Both are + * built at create time because which one a draw uses is not known until the + * draw - a shader whose uniforms do not fit the primary attribute bank has to + * go back to the draw module. */ +struct sgx_pipe_vs { + void *draw_vs; + struct sgx_shader *hw; + /* The same program with its colour output packed into one dword, for + * a fragment program that takes a packed colour. The transform cannot + * write both forms, and which is wanted belongs to the pair, so both + * are compiled and the bind picks one. */ + struct sgx_shader *hw_packed; + /* And the same program numbered for the frame a particular fragment + * program builds: the frame numbers sampled sets first, so which slot + * a varying belongs in is a property of the pair and is not known when + * the vertex shader is created. Compiled when the two are first bound + * together and kept while the pairing holds. src is the shader as it + * arrived, to compile from again - only a cloned NIR, because the + * caller's tokens are freed once create returns. */ + struct pipe_shader_state src; + int have_src; + /* Kept rather than replaced: a frame already submitted can still be + * reading the one a new pairing would displace, and freeing it under + * the X server is a use-after-free that took it down. Grown as the + * client cycles through pairings - a fixed four meant ioquake3 ran out + * and fell back to a form compiled for another layout, which fetched + * the record at a stride nothing had written and faulted the MMU. */ + struct sgx_shader **hw_var; + unsigned *var_key; + unsigned nvar, avar; +}; + +struct sgx_pipe_fence { + struct pipe_reference reference; +}; + +/* Mesa's format enumeration onto the hardware's. Only the formats the + * hardware has a code for appear; everything else is PIPE_FORMAT_NONE's + * counterpart, SGX_FMT_NONE, and is refused rather than approximated. */ +static enum sgx_format sgx_format_of(enum pipe_format f) +{ + switch (f) { + case PIPE_FORMAT_A8_UNORM: return SGX_FMT_A8; + /* One 8-bit channel, which gl-re/textures.md has serving a8, l8 and + * i8 alike - so the same code answers for red. The X server wants a + * depth-eight pixmap it can render into for its glyph atlas, and a + * GLES 2 context takes a red one as a colour attachment where it + * refuses an alpha one. */ + case PIPE_FORMAT_R8_UNORM: return SGX_FMT_A8; + /* The same byte again for luminance and intensity: the fetch hands + * it back in alpha and sgx_set_unit_swizzle() points every channel + * the view names at that byte. Without these rows mesa/st stored a + * GL_LUMINANCE texture as B8G8R8X8, four times the memory. */ + case PIPE_FORMAT_L8_UNORM: return SGX_FMT_A8; + case PIPE_FORMAT_I8_UNORM: return SGX_FMT_A8; + case PIPE_FORMAT_R8G8_UNORM: return SGX_FMT_AL88; + case PIPE_FORMAT_B4G4R4A4_UNORM: return SGX_FMT_A4R4G4B4; + case PIPE_FORMAT_B5G5R5A1_UNORM: return SGX_FMT_A1R5G5B5; + case PIPE_FORMAT_B5G6R5_UNORM: return SGX_FMT_R5G6B5; + case PIPE_FORMAT_B8G8R8A8_UNORM: return SGX_FMT_A8R8G8B8; + case PIPE_FORMAT_B8G8R8X8_UNORM: return SGX_FMT_A8R8G8B8; + case PIPE_FORMAT_R8G8B8A8_UNORM: return SGX_FMT_A8B8G8R8; + /* The part has no sRGB code: the bytes are stored and fetched as + * plain 8888, and the view's GAMMA bit has the unit decode them + * (sgx_srgb_hw()); the fragment program's own decode (sgx_shader.c) + * is the fallback. */ + case PIPE_FORMAT_B8G8R8A8_SRGB: return SGX_FMT_A8R8G8B8; + case PIPE_FORMAT_R8G8B8A8_SRGB: return SGX_FMT_A8B8G8R8; + case PIPE_FORMAT_YUYV: return SGX_FMT_YUY2; + case PIPE_FORMAT_UYVY: return SGX_FMT_UYVY; + /* The float and 16-bit family, sampled only. Each is the code of + * one plane; sgx_format_chunks() says how many planes a texel is + * spread over, as sgx535pixfmts.h lays them out. The one-channel + * ones come back in red, the view swizzle mesa/st composes for + * luminance, alpha and intensity does the rest. */ + case PIPE_FORMAT_R32_FLOAT: + case PIPE_FORMAT_L32_FLOAT: + case PIPE_FORMAT_A32_FLOAT: + case PIPE_FORMAT_I32_FLOAT: + case PIPE_FORMAT_R32G32_FLOAT: + case PIPE_FORMAT_R32G32B32A32_FLOAT: return SGX_FMT_F32; + case PIPE_FORMAT_R16_FLOAT: + case PIPE_FORMAT_L16_FLOAT: + case PIPE_FORMAT_A16_FLOAT: + case PIPE_FORMAT_I16_FLOAT: return SGX_FMT_F16; + case PIPE_FORMAT_R16G16_FLOAT: + case PIPE_FORMAT_R16G16B16A16_FLOAT: return SGX_FMT_F1616; + case PIPE_FORMAT_R16_UNORM: + case PIPE_FORMAT_L16_UNORM: + case PIPE_FORMAT_A16_UNORM: + case PIPE_FORMAT_I16_UNORM: return SGX_FMT_U16; + case PIPE_FORMAT_R16_SNORM: return SGX_FMT_S16; + case PIPE_FORMAT_R16G16_UNORM: + case PIPE_FORMAT_R16G16B16A16_UNORM: return SGX_FMT_U1616; + case PIPE_FORMAT_R16G16_SNORM: + case PIPE_FORMAT_R16G16B16A16_SNORM: return SGX_FMT_S1616; + /* ETC1: the TAG decodes it to a packed 8888 texel. */ + case PIPE_FORMAT_ETC1_RGB8: return SGX_FMT_ETC1; + default: return SGX_FMT_NONE; + } +} + +/* How many planes a texel of this format is stored in, each a texture of + * sgx_format_of()'s code: the vendor's ui32NumChunks. */ +static unsigned sgx_format_chunks(enum pipe_format f) +{ + switch (f) { + case PIPE_FORMAT_R32G32_FLOAT: + case PIPE_FORMAT_R16G16B16A16_FLOAT: + case PIPE_FORMAT_R16G16B16A16_UNORM: + case PIPE_FORMAT_R16G16B16A16_SNORM: return 2; + case PIPE_FORMAT_R32G32B32A32_FLOAT: return 4; + default: return 1; + } +} + +/* The code a sampler view of the format is described with. A depth resource + * has no code of its own - the frame owns the surface - and the texture unit + * has no depth format at any width: EURASIA_PDS_DOUTT1_TEXFORMAT_* runs from + * U8 to F32 with no depth row, and the two that would serve one, X8U24 (21) + * and U8U24 (22), are behind SGX545/543/544/554 (sgxdefs.h:4607-4619). + * + * What the surface holds is not a float either. The ISP's store format is + * I24ZI8S - ZLSCTL [19:18] = 1, the 0x00450000 the raster stream carries + * (xpsb_frame.c:2887) - a 24-bit integer depth with the stencil byte above + * it, measured at a window depth of 0.25 as the 0x00400000 recorded in + * xpsb_frame.c:3129-3140. Read as F32 that is a denormal, which is why every + * depth sample came back as zero. + * + * So it is described as the 32-bit BGRA it is, which is the vendor's own row + * for D24X8 and D24S8 (sgx535pixfmts.h:758, 774), and the fragment program + * puts the three bytes back together - sgx_depth_expand() in sgx_shader.c. */ +static enum sgx_format sgx_view_format_of(enum pipe_format f) +{ + if (util_format_is_depth_or_stencil(f)) + return SGX_FMT_A8R8G8B8; + return sgx_format_of(f); +} + +/* What a depth surface holds, for the CPU: the ISP's I24ZI8S store format is + * a little-endian dword with the depth in [23:0] and the stencil above it, + * which is exactly this. The texture unit is given the BGRA code instead - + * it has no depth format - but every conversion the CPU does goes through + * this one. */ +#define SGX_DEPTH_SURF_FMT PIPE_FORMAT_Z24_UNORM_S8_UINT + +/* Whether a view is sampled through that reassembly, which is every depth + * one: the program carries it, so a change of which units hold one rebuilds + * the program the way an sRGB view does. */ +static int sgx_depth_in_shader(const struct pipe_sampler_view *pv) +{ + return pv && util_format_is_depth_or_stencil(pv->format); +} + +static unsigned sgx_bind_of(unsigned pipe_bind) +{ + unsigned b = 0; + + if (pipe_bind & PIPE_BIND_SAMPLER_VIEW) + b |= SGX_BIND_SAMPLER_VIEW; + if (pipe_bind & (PIPE_BIND_RENDER_TARGET | PIPE_BIND_DISPLAY_TARGET | + PIPE_BIND_SCANOUT)) + b |= SGX_BIND_RENDER_TARGET; + if (pipe_bind & PIPE_BIND_VERTEX_BUFFER) + b |= SGX_BIND_VERTEX_BUFFER; + if (pipe_bind & PIPE_BIND_INDEX_BUFFER) + b |= SGX_BIND_INDEX_BUFFER; + return b; +} + +/* Gallium's comparison functions are in the same order as the hardware's, + * which is worth asserting rather than relying on: they are separate + * enumerations that happen to agree today. */ +static enum sgx_func sgx_func_of(unsigned pipe_func) +{ + switch (pipe_func) { + case PIPE_FUNC_NEVER: return SGX_FUNC_NEVER; + case PIPE_FUNC_LESS: return SGX_FUNC_LESS; + case PIPE_FUNC_EQUAL: return SGX_FUNC_EQUAL; + case PIPE_FUNC_LEQUAL: return SGX_FUNC_LEQUAL; + case PIPE_FUNC_GREATER: return SGX_FUNC_GREATER; + case PIPE_FUNC_NOTEQUAL: return SGX_FUNC_NOTEQUAL; + case PIPE_FUNC_GEQUAL: return SGX_FUNC_GEQUAL; + case PIPE_FUNC_ALWAYS: + default: return SGX_FUNC_ALWAYS; + } +} + +static enum sgx_stencil_op sgx_stencil_op_of(unsigned op) +{ + switch (op) { + case PIPE_STENCIL_OP_KEEP: return SGX_STENCIL_KEEP; + case PIPE_STENCIL_OP_ZERO: return SGX_STENCIL_ZERO; + case PIPE_STENCIL_OP_REPLACE: return SGX_STENCIL_REPLACE; + case PIPE_STENCIL_OP_INCR: return SGX_STENCIL_INCR; + case PIPE_STENCIL_OP_DECR: return SGX_STENCIL_DECR; + case PIPE_STENCIL_OP_INCR_WRAP: return SGX_STENCIL_INCR_WRAP; + case PIPE_STENCIL_OP_DECR_WRAP: return SGX_STENCIL_DECR_WRAP; + case PIPE_STENCIL_OP_INVERT: + default: return SGX_STENCIL_INVERT; + } +} + +static unsigned sgx_pipe_float_comps_fmt(unsigned f, int *half); + +static bool sgx_pipe_format_answer(struct pipe_screen *ps, + enum pipe_format format, + enum pipe_texture_target target, + unsigned sample_count, + unsigned storage_sample_count, + unsigned bind) +{ + enum sgx_format f = sgx_format_of(format); + unsigned need = 0; + + (void)ps; + /* Multisampling: one sample a pixel, or the part's 2x2 - and nothing + * between them. The vendor's SGXAddRenderTarget() accepts 1x1, 2x1, + * 1x2 and 2x2 (sgxrender_targets.c:738-753), but the two two-sample + * forms are behind SGX_FEATURE_MSAA_2X_IN_X and _IN_Y and the SGX535 + * block of sgxfeaturedefs.h (189-232) defines neither. So two samples + * a pixel is refused here rather than rounded up: mesa/st probes + * downwards from its own maximum and takes the first count that is + * accepted, so refusing two is what makes it settle on four. + * + * Storage is either the same count or one: the part holds the samples + * in the tile buffer and the pixel back end downscales them on store, + * so a multisampled colour surface is single-sampled memory. That is + * what pipe_caps.surface_sample_count describes, and it is why + * EXT_multisampled_render_to_texture is the natural interface to this + * hardware rather than a resolve blit. */ + if (sample_count > 1) { + if (!xpsb_msaa_axis(sample_count)) + return false; + if (storage_sample_count > 1 && + storage_sample_count != sample_count) + return false; + /* Only a render target or a depth attachment is ever + * multisampled here: nothing samples a multisample texture, + * because the descriptor has no per-sample form. */ + if (bind & ~(unsigned)(PIPE_BIND_RENDER_TARGET | + PIPE_BIND_DEPTH_STENCIL)) + return false; + } else if (storage_sample_count > 1) { + return false; + } + /* A cube and a volume are texture types of their own in the + * descriptor, so they are described rather than refused. A 1D + * texture is a 2D one a row high - the unit has no 1D type, and the + * sample is lowered to a 2D one in sgx_pipe_create_shader(). */ + /* An array texture has no descriptor on this part and cannot be given + * one. TEXTYPE, DOUTT1 [31:29], holds 2D, 3D, CEM, STRIDE and TILED + * on this core and nothing else (sgxdefs.h:4558-4567; the ARB_* codes + * at :4549-4556 are arbitrary sizes, not arrays, and are behind + * SGX_FEATURE_TAG_NPOT_TWIDDLE, which 535 does not have). No SGX core + * in the DDK defines an array type at all - the one ARRAY name in + * hwdefs is SGX_FEATURE_RENDER_TARGET_ARRAYS, 545's layered render + * targets (sgxfeaturedefs.h:509). There is no layer operand either: + * SMP's coordinate count stops at CDIM_UVS and the fourth encoding is + * reserved (sgxdefs.h:6332-6337), and this core has three texture + * control words, not the four that carry a linear DEPTH elsewhere + * (EURASIA_TAG_TEXTURE_STATE_SIZE 3, sgxdefs.h:5047-5053). + * + * A volume is not a substitute. Its depth is log2 only (SSIZE, + * sgxdefs.h:4730-4744) and there is one filter for all three axes - + * MINFILTER/MAGFILTER/MIPFILTER, sgxdefs.h:4391-4405 - so a linear + * filter blends slices, and the only way to stop that also gives up + * bilinear inside the layer. + * + * The vendor's own answer, in its D3D compiler and never in GL, is to + * emulate: the shader computes TEX_BASE = ARRAY_BASE + ARRAY_INDEX * + * ELEMENT_SIZE and writes it into the descriptor's address word + * before issuing the sample (usc2/icvt_core.c:3826-3844), which + * forces the sample dependent (regpack.c:5666-5675), turns on + * per-instance mode (icvt_core.c:5771-5777) and cannot be + * predicated. That is a compiler feature, not a descriptor one, and + * it is out of reach of a driver whose descriptors are built by the + * PDS primary program. So this is refused, not approximated. */ + if (target != PIPE_TEXTURE_2D && target != PIPE_TEXTURE_RECT && + target != PIPE_BUFFER && target != PIPE_TEXTURE_CUBE && + target != PIPE_TEXTURE_1D && target != PIPE_TEXTURE_3D) + return false; + /* A volume is sampled and nothing else: every 3D encoding is a + * twiddled one, which the raster pass cannot write, and the packed + * YUV pair has no twiddled form at all. */ + if (target == PIPE_TEXTURE_3D && + ((bind & ~(unsigned)PIPE_BIND_SAMPLER_VIEW) || + f == SGX_FMT_YUY2 || f == SGX_FMT_UYVY)) + return false; + /* The depth buffer is the frame's own, four bytes a pixel and never + * described by a surface descriptor, so it has no code in the format + * table - it is answered for here rather than being refused as an + * unknown format. */ + /* NOT refused for sampling, though GL says a texture with no alpha + * channel samples alpha as one and this part stores such a format in + * one that has alpha, so the stored byte comes back instead. Refusing + * it here - to make the state tracker pick an alpha format - breaks + * the X server outright: glamor samples the BGRX pixmaps it composites + * every window from, and they all came out as noise. Forcing the one + * in the fragment program instead costs a rebuild whenever a view + * changes, which for X is constantly: 2.4 fps and fifty stalls. + * + * So the contract is knowingly unmet for now, as it was before. It + * shows up as ioquake3's intro cinematic blending to nothing (an RGB8 + * video sampled for its alpha). The fix belongs at upload - filling + * the unused byte with 0xff when the caller's format has no alpha - + * which costs nothing at draw time; the path Mesa actually uploads + * these through has yet to be found (not texture_subdata, not + * texture_unmap). */ + if (util_format_is_depth_or_stencil(format)) + return (bind & ~(PIPE_BIND_DEPTH_STENCIL | PIPE_BIND_SAMPLER_VIEW)) == 0 && + (format == PIPE_FORMAT_Z24X8_UNORM || + format == PIPE_FORMAT_Z24_UNORM_S8_UINT || + format == PIPE_FORMAT_Z16_UNORM); + /* Vertex and index data, which nothing here used to answer for: every + * bind reaching this point without a sampler or render bit fell out + * with need == 0 and a flat false. That is not "the part cannot read + * it" but "nobody was asked" - u_vbuf_get_caps() takes a false here + * as "translate this array before every draw", and one false is + * enough to set fallback_always, so the driver was paying a CPU + * rewrite of every vertex array in every frame including the plain + * float ones. The formats named are the ones the vertex path decodes: + * one to four floats at 32 or 16 bits. SGX_NO_VTX_FORMATS restores + * the blanket refusal. */ + if (bind & (PIPE_BIND_VERTEX_BUFFER | PIPE_BIND_INDEX_BUFFER)) { + if (bind & ~(unsigned)(PIPE_BIND_VERTEX_BUFFER | + PIPE_BIND_INDEX_BUFFER)) + return false; + if (SGX_ENVS("SGX_NO_VTX_FORMATS")) + return false; + if (bind & PIPE_BIND_INDEX_BUFFER) + /* One, two and four bytes, which is what both the + * draw module and sgx_hwtcl_index() already read. */ + return format == PIPE_FORMAT_R8_UINT || + format == PIPE_FORMAT_R16_UINT || + format == PIPE_FORMAT_R32_UINT; + if (sgx_pipe_float_comps_fmt((unsigned)format, NULL)) + return 1; + /* The draw module fetches these itself. Refusing them made + * u_vbuf translate every vertex of every draw on the CPU + * instead - unpacking each component to float, and scanning + * the index buffer for its range to know what to translate. + * ioquake3 keeps its normals and tangents as normalised + * shorts, so that was four per cent of the process. The + * part's own fetch reads none of them, which is why + * sgx_hwtcl_can() asks the narrower question. */ + return format == PIPE_FORMAT_R16_UNORM || + format == PIPE_FORMAT_R16G16_UNORM || + format == PIPE_FORMAT_R16G16B16_UNORM || + format == PIPE_FORMAT_R16G16B16A16_UNORM || + format == PIPE_FORMAT_R16_SNORM || + format == PIPE_FORMAT_R16G16_SNORM || + format == PIPE_FORMAT_R16G16B16_SNORM || + format == PIPE_FORMAT_R16G16B16A16_SNORM || + format == PIPE_FORMAT_R8G8B8A8_UNORM || + format == PIPE_FORMAT_B8G8R8A8_UNORM || + format == PIPE_FORMAT_R8G8B8A8_SNORM || + format == PIPE_FORMAT_R8_UNORM || + format == PIPE_FORMAT_R8G8_UNORM; + } + if (f == SGX_FMT_NONE) + return false; + + if (bind & PIPE_BIND_SAMPLER_VIEW) + need |= SGX_FMT_BIND_SAMPLER; + if (bind & (PIPE_BIND_RENDER_TARGET | PIPE_BIND_DISPLAY_TARGET | + PIPE_BIND_SCANOUT)) { + /* The frame programs the target's pixel-format field now, so + * the depth the caller asks for is the depth that comes out. + * The other channel order is still withheld: offering it makes + * mesa/st choose it and every red pixel comes out blue, which + * is a separate thing from the depth. + * + */ + if (format != PIPE_FORMAT_B8G8R8A8_UNORM && + format != PIPE_FORMAT_B8G8R8X8_UNORM && + format != PIPE_FORMAT_B5G6R5_UNORM && + format != PIPE_FORMAT_B5G5R5A1_UNORM && + format != PIPE_FORMAT_R8_UNORM && + format != PIPE_FORMAT_R8G8_UNORM) + return false; + need |= SGX_FMT_BIND_RENDER; + } + if (!need) + return false; + return sgx_format_supported(f, need) != 0; +} + +/* pipe_screen::is_format_supported. Mesa asserts on a texture format it was + * told is unsupported rather than falling back, so a refusal here is a crash + * later; SGX_FMT_TRACE names the one that was refused. */ +static bool sgx_pipe_is_format_supported(struct pipe_screen *ps, + enum pipe_format format, + enum pipe_texture_target target, + unsigned sample_count, + unsigned storage_sample_count, + unsigned bind) +{ + bool ok = sgx_pipe_format_answer(ps, format, target, sample_count, + storage_sample_count, bind); + + if (!ok && SGX_ENVS("SGX_FMT_TRACE")) + fprintf(sgx_log(), "sgx: format %s refused for target %d, " + "%u sample(s), %u storage, bind 0x%x\n", + util_format_short_name(format), (int)target, + sample_count, storage_sample_count, bind); + return ok; +} + +static const char *sgx_pipe_get_name(struct pipe_screen *ps) +{ + (void)ps; + return "SGX535"; +} + +static const char *sgx_pipe_get_vendor(struct pipe_screen *ps) +{ + (void)ps; + return "Imagination Technologies"; +} + +static const char *sgx_pipe_get_device_vendor(struct pipe_screen *ps) +{ + (void)ps; + return "Intel"; +} + +/* Map the caps that have a measured answer. Mesa's default for anything not + * set here is zero, which is the honest answer for a capability this driver + * does not have. */ +/* What the hardware can do, in Gallium's terms. + * + * The vertex stage runs in the draw module - the hardware's vertex program is + * still the frame's pass-through, so a caller's vertex shader is executed on + * the CPU and what reaches the tiler is already in screen space. That is why + * the vertex shader caps come from draw_init_shader_caps() rather than from + * anything measured here: they describe draw's interpreter, which is what will + * actually run the program. + * + * The fragment caps are the hardware's, and every one of them carries where it + * came from in sgx_screen.c. + */ +static void sgx_pipe_init_shader_caps(struct pipe_screen *ps) +{ + struct pipe_shader_caps *c = + (struct pipe_shader_caps *)&ps->shader_caps[MESA_SHADER_VERTEX]; + + draw_init_shader_caps(c); + c->supported_irs = (1 << PIPE_SHADER_IR_NIR) | (1 << PIPE_SHADER_IR_TGSI); + /* mesa/st insists the two stages agree, and the fragment pipeline has + * no integer path at all. */ + c->integers = false; + c->int16 = false; + c->fp16 = false; + c->fp16_derivatives = false; + c->fp16_const_buffers = false; + c->glsl_16bit_load_dst = false; + c->indirect_temp_addr = false; + /* Vertex texture fetch, which the draw module does on the CPU - + * the part has no vertex texture unit. Reported as zero before, + * which is what made glmark2's terrain refuse to run at all: + * it wants GL_MAX_VERTEX_TEXTURE_IMAGE_UNITS above zero and + * says "Unsupported" without it. */ + c->max_texture_samplers = getenv("SGX_NO_VTF") ? 0u : + SGX_MAX_VTX_SAMPLERS; + c->max_sampler_views = c->max_texture_samplers; + c->max_shader_buffers = 0; + c->max_shader_images = 0; + + c = (struct pipe_shader_caps *)&ps->shader_caps[MESA_SHADER_FRAGMENT]; + c->supported_irs = (1 << PIPE_SHADER_IR_NIR) | (1 << PIPE_SHADER_IR_TGSI); + c->max_instructions = 512; + c->max_alu_instructions = 512; + c->max_tex_instructions = 512; + c->max_tex_indirections = 8; + c->max_control_flow_depth = 8; + c->max_inputs = 8; + /* Two: the colour, and a dual source blend's second colour. The second + * is a temporary the blend reads and never a pixel, but the program + * declares it as an output and the count has to say so. */ + c->max_outputs = 2; + /* What a task can name, less what a blend appended to the program + * needs to stage: the constant, a pre-scaled destination and a masked + * result, plus the temporary the closing move is redirected into. + * Thirty-two was the old five-bit field's limit and outlived it - the + * count rides in two DOUTU words now and sgx_fs_fits() has been + * letting programs through at ninety-six all along. */ + c->max_temps = sgx_temp_max() - SGX_BLEND_TEMP_HEADROOM; + /* The fragment stage's constants are not the vertex stage's, and this + * used to be given the vertex figure: eighteen dwords, which is 72 + * components, less than half what GL asks for. + * + * The bank is the limit, not the compiler's literal pool. A fragment + * program's constants are DMAed into the secondary attribute bank, so + * a caller allowed 64 vec4 could ask for 256 registers of a bank that + * holds 128 - and did: glamor's composite arrives with 46 uniform + * quads and overflows in codegen rather than being told a limit. + * Thirty-two vec4 is what the bank holds and is twice what GL 2.1 and + * GLES2 require. Samplers and literals come out of the same bank, so + * a program combining the full allowance with either is still refused + * - by sgx_fs_fits(), which says so. */ + c->max_const_buffer0_size = sgx_sa_max() * sizeof(float); + c->max_const_buffers = 1; + /* The part's eight, which is what the vendor's own compiler serves. + * SGX_SAMPLERS_SERVED reports only what the frame can issue instead - + * fewer, while the primary PDS data segment carries three - which is + * honest but stops ioquake3's opengl2 renderer dead: lightall_fp.glsl + * declares seven samplers, the link fails, and the client exits before + * it draws anything. A unit the frame cannot issue reports itself and + * samples black, so the shortfall is visible rather than silent. */ + { + unsigned nu = (unsigned)sgx_screen_cap(SGX_CAP_MAX_TEXTURE_UNITS); + + if (SGX_ENVS("SGX_SAMPLERS_SERVED") && nu > SGX_MAX_TEX_UNITS) + nu = SGX_MAX_TEX_UNITS; + if (nu > SGX_MAX_SAMPLERS) + nu = SGX_MAX_SAMPLERS; + c->max_texture_samplers = nu; + c->max_sampler_views = nu; + } + /* The parser refuses indirect addressing and the compiler has no form + * for it, so claiming it leaves the indirection in the TGSI, the + * shader is refused, and every draw using it is silently skipped - + * which is why a uniform array renders nothing while the same shader + * with named uniforms is correct. Declining it makes the state tracker + * lower the indexing into something this driver can compile. */ + c->indirect_const_addr = false; + c->tgsi_any_inout_decl_range = true; +} + +/* The fragment stage takes NIR and hands it to nir_to_tgsi, so the NIR it can + * accept is whatever that pass can lower; the vertex stage runs in the draw + * module's interpreter and takes what that can. Both sets are the ones i915 + * uses, which is the driver in the tree that is in the same position: no + * integers, TGSI underneath, and the vertex stage in software. + */ +static const nir_shader_compiler_options sgx_fs_nir_options = { + .fdot_replicates = true, + .float_mul_add32 = nir_float_muladd_support_has_fmad | nir_float_muladd_support_fuse, + .lower_bitops = true, /* nir_to_tgsi needs it without integers */ + .lower_extract_byte = true, + .lower_extract_word = true, + .lower_fdiv = true, + .lower_fdph = true, + .lower_flrp32 = true, + .lower_fmod = true, + .lower_sincos = true, + .lower_uniforms_to_ubo = true, + .force_indirect_unrolling = nir_var_all, + .force_indirect_unrolling_sampler = true, + .max_unroll_iterations = 32, + .no_integers = true, + .has_fused_comp_and_csel = true, +}; + +static const struct nir_shader_compiler_options sgx_vs_nir_options = { + .lower_bitops = true, /* nir_to_tgsi needs it without integers */ + .lower_scmp = true, + .lower_flrp32 = true, + .lower_flrp64 = true, + .lower_fsat = true, + .lower_bitfield_insert = true, + .lower_bitfield_extract = true, + .lower_fdph = true, + .lower_fmod = true, + .lower_hadd = true, + .lower_uadd_sat = true, + .lower_usub_sat = true, + .lower_iadd_sat = true, + .lower_pack_snorm_2x16 = true, + .lower_pack_snorm_4x8 = true, + .lower_pack_unorm_2x16 = true, + .lower_pack_unorm_4x8 = true, + .lower_pack_half_2x16 = true, + .lower_pack_split = true, + .lower_unpack_snorm_2x16 = true, + .lower_unpack_snorm_4x8 = true, + .lower_unpack_unorm_2x16 = true, + .lower_unpack_unorm_4x8 = true, + .lower_unpack_half_2x16 = true, + .lower_extract_byte = true, + .lower_extract_word = true, + .lower_uadd_carry = true, + .lower_usub_borrow = true, + .lower_mul_2x32_64 = true, + .lower_ifind_msb = true, + .max_unroll_iterations = 32, + .lower_cs_local_index_to_id = true, + .lower_uniforms_to_ubo = true, + .lower_device_index_to_zero = true, + /* .support_16bit_alu = true, */ + .support_indirect_inputs = (uint8_t)BITFIELD_MASK(MESA_SHADER_STAGES), + .support_indirect_outputs = (uint8_t)BITFIELD_MASK(MESA_SHADER_STAGES), + .no_integers = true, + .has_fused_comp_and_csel = true, +}; + + +static void sgx_pipe_init_caps(struct pipe_screen *ps) +{ + /* pipe_screen::caps is const to callers; a driver fills it in at + * creation, which is what the cast is for and the only place it is + * legitimate. */ + struct pipe_caps *c = (struct pipe_caps *)(uintptr_t)&ps->caps; + + u_init_pipe_screen_caps(ps, 1); + + c->max_render_targets = sgx_screen_cap(SGX_CAP_MAX_RENDER_TARGETS); + /* One, which is GL_ARB_blend_func_extended (st_extensions.c:1530). + * The part has no fixed-function blend unit to lack the feature: + * blending is SOP instructions at the end of the fragment program, so + * the second colour is a register that program wrote and the blend + * reads it as SOP3's src0. That slot also carries the blend constant, + * so a state wanting both is refused - sgx_blend_build(). */ + c->max_dual_source_render_targets = + SGX_ENVS("SGX_NO_DUAL_SRC") ? 0 : 1; + /* Sampling what an earlier draw wrote, without a copy in between. + * sgx_pipe_texture_barrier() ends the frame, which is what makes the + * writes visible here; the X server uses it to copy inside one pixmap + * without allocating a temporary to bounce through + * (glamor_copy_needs_temp). */ + c->texture_barrier = !SGX_ENVS("SGX_NO_TEX_BARRIER"); + c->npot_textures = sgx_screen_cap(SGX_CAP_NPOT_TEXTURES) != 0; + /* The part counts: eight ISP visibility registers, armed per object + * through ISP word B (sgxdefs.h:1435-1438), read back by the kernel + * when the render ends. PIPE_QUERY_OCCLUSION_COUNTER is that count and + * both predicate forms are it tested against zero; every other query + * type is refused by name in sgx_pipe_create_query(). This one cap + * governs both GL_SAMPLES_PASSED and GL_ANY_SAMPLES_PASSED + * (st_extensions.c:1061 gates ARB_occlusion_query2 on it). */ + c->occlusion_query = sgx_screen_cap(SGX_CAP_OCCLUSION_QUERY) != 0; + /* The part has DSX and DSY - group 0x04, corroborated by Imagination's + * own compiler emitting fdsx for dFdx - and the backend emits them + * now, so the extension that lets a GLES2 shader ask for them is + * advertised. Desktop GLSL has dFdx in core and reached them either + * way. */ + c->fragment_shader_derivatives = true; + /* Nothing here reads pipe_depth_stencil_alpha_state's alpha fields, so + * claiming the test drew the fragments it should have discarded. + * Declining it makes the state tracker lower the test into the shader + * as a discard, which this driver does implement. */ + c->alpha_test = false; + /* Without this the state tracker never installs GetGraphicsResetStatus + * and sgx_pipe_get_device_reset_status() is never called, so a lost + * render stays invisible to the application. */ + c->device_reset_status_query = true; + /* The view's swizzle is already carried in the fragment program - + * sgx_set_unit_swizzle() and sgx_shader_retarget_swz() - because the + * one-byte formats need it. A GL_TEXTURE_SWIZZLE_* is composed into + * the same view swizzle by mesa/st, so it costs nothing new except a + * program variant per distinct swizzle. */ + c->texture_swizzle = true; + /* Without this Mesa never calls the driver's own reduction and falls + * back to rendering into each level, which a twiddled surface cannot + * be a render target for. */ + c->generate_mipmap = true; + c->max_texture_2d_size = + (unsigned)sgx_screen_cap(SGX_CAP_MAX_TEXTURE_SIZE); + /* The unit describes a cube face, and every axis of a volume, by log2 + * of one side, four bits, so each may be 4096 - the same ceiling a 2D + * texture has. The levels count is that log2 plus one. + * + * This and sgx_pipe_is_format_supported() have to agree: a non-zero + * count with the format refused has mesa/st advertise GL_TEXTURE_3D + * and then assert in st_texture_create() on the refusal it was just + * given. Both say yes now. The slice reaches the unit through the USE + * rather than the iterator - sgx_shader.c refuses the iterated form + * for a volume or cube sample, because the iterator hands this core + * two coordinates and a projection and the slice never arrived under + * TEXPROJ_T or RHW. */ + c->max_texture_3d_levels = + util_logbase2(sgx_screen_cap(SGX_CAP_MAX_TEXTURE_SIZE)) + 1; + c->max_texture_cube_levels = + util_logbase2(sgx_screen_cap(SGX_CAP_MAX_TEXTURE_SIZE)) + 1; + /* Zero, and said rather than left to the struct: the descriptor has no + * array texture type at all - see sgx_pipe_is_format_supported() - + * and this cap is what EXT_texture_array is gated on + * (st_extensions.c:1123) as well as GL_MAX_ARRAY_TEXTURE_LAYERS. */ + c->max_texture_array_layers = 0; + c->vertex_color_unclamped = true; + c->mixed_colorbuffer_formats = false; + + /* GLSL 1.20 is what a driver without integers can honestly claim, and + * it is what ES 2.0 needs. */ + c->glsl_feature_level = 120; + c->glsl_feature_level_compatibility = 120; + + /* All from the draw module, which is what runs the vertex stage. */ + /* Not while the threaded context is in circuit: a user buffer is a + * pointer into the caller's memory, and the caller is free to write + * over it as soon as the call returns - which under a deferred driver + * thread is long before the draw is executed. Mesa uploads them for + * us instead. */ + c->user_vertex_buffers = !SGX_ENVS("SGX_THREAD"); + c->vertex_element_instance_divisor = true; + c->primitive_restart = true; + c->primitive_restart_fixed_index = true; + c->vs_instanceid = true; + c->vertex_color_clamped = true; + c->shareable_shaders = false; /* they hold draw's per-context state */ + + /* The samples live in the tile buffer and the pixel back end + * downscales them on store, so a multisampled render writes a + * single-sampled surface and there is no resolve for the driver to + * do. That is exactly what this cap declares, and it is what gives + * ES 2 GL_EXT_multisampled_render_to_texture on this part. */ + c->surface_sample_count = true; + c->mixed_framebuffer_sizes = true; + c->mixed_color_depth_bits = true; + c->blend_equation_separate = true; + c->tgsi_texcoord = true; + c->max_varyings = 8; + c->max_vertex_attrib_stride = 2048; + c->max_viewports = 1; + c->constant_buffer_offset_alignment = 16; + /* Constants in a real buffer rather than a pointer into the state + * tracker's parameter list. The draw module's JIT bounds-checks every + * constant read against the size it was given and returns zero past + * it, and on the user-buffer path that size is the parameter list's + * own byte count - short of what the shader reads, so a projection + * matrix came back as zeros, w as zero, and every vertex through the + * module as -nan. llvmpipe, which drives the same JIT, sets this and + * asserts a user buffer never reaches it. */ + c->prefer_real_buffer_in_constbuf0 = true; + c->min_map_buffer_alignment = 64; + c->fs_coord_origin_upper_left = true; + c->fs_coord_pixel_center_half_integer = true; + c->endianness = PIPE_ENDIAN_LITTLE; + c->uma = true; + c->vendor_id = 0x8086; + c->device_id = 0x8108; + + c->min_line_width = 1; + c->min_line_width_aa = 1; + c->min_point_size = 1; + c->min_point_size_aa = 1; + c->line_width_granularity = 0.1f; + c->point_size_granularity = 0.1f; + /* What the hardware rasterises, when it is doing the rasterising: the + * ISP's width field is four bits as width - 1, so sixteen pixels, and + * a point may be up to EURASIA_MAX_POINT_SIZE. The CPU widening path + * has no such limit, but claiming more than the part can draw would + * be a promise the native path cannot keep. Antialiased points and + * lines are not supported at all - the vendor reports 1.0 for both - + * so the _aa caps say so rather than repeating the aliased figure. */ + c->max_line_width = (float)SGX_ISP_PLWIDTH_MAX; + c->max_line_width_aa = 1.0f; + c->max_point_size = 511.0f; + c->max_point_size_aa = 1.0f; + /* The unit's ratio field reaches 16x, but only claim it where the + * driver will actually program it - see sgx_aniso_enabled(). The + * ratio alone does not expose the extension: Mesa gates that on the + * boolean beside it, so both have to move together. */ + c->anisotropic_filter = sgx_aniso_enabled(); + c->max_texture_anisotropy = sgx_aniso_enabled() ? + SGX_MAX_ANISOTROPY : 1.0f; + /* The two clamp modes Mesa will not reach without being told. Left + * unset it refuses GL_MIRROR_CLAMP_TO_EDGE before a sampler state is + * ever built, so the translation below would be unreachable. */ + c->texture_mirror_clamp = true; + c->texture_mirror_clamp_to_edge = true; + /* Not a claim that the unit has GL_CLAMP's own mode - it has one and + * it does not work, see sgx_wrap_of(). Left unset, Mesa emulates + * GL_CLAMP by rewriting the fragment shader, and that variant renders + * nothing here; claiming the cap keeps the draw on the ordinary clamp, + * which is correct wherever no border colour is involved. */ + c->gl_clamp = true; + /* What the unit's adjust field actually reaches, not a round number: + * it is six bits of signed 3.3 fixed point, so +4.0 is the largest + * bias it can express (XPSB_LOD_BIAS_MAX). Reporting sixteen promised + * a range three quarters of which silently clamped. */ + c->max_texture_lod_bias = XPSB_LOD_BIAS_MAX; +} + +static const void *sgx_pipe_resource_cpu(struct pipe_context *pc, + struct pipe_resource *pres); +/* Mark a buffer's cached copy out of date; see sgx_pipe_resource_cpu_cached(). */ +static void sgx_pipe_shadow_dirty(struct pipe_resource *pres); + +/* --- resources ---------------------------------------------------------- */ + +static unsigned sgx_bind_of_pipe(unsigned bind) +{ + unsigned out = 0; + + if (bind & PIPE_BIND_SAMPLER_VIEW) + out |= SGX_BIND_SAMPLER_VIEW; + if (bind & (PIPE_BIND_RENDER_TARGET | PIPE_BIND_DISPLAY_TARGET | + PIPE_BIND_SCANOUT | PIPE_BIND_SHARED)) + out |= SGX_BIND_RENDER_TARGET; + if (bind & (PIPE_BIND_SCANOUT | PIPE_BIND_DISPLAY_TARGET | + PIPE_BIND_SHARED)) + out |= SGX_BIND_SCANOUT; + if (bind & PIPE_BIND_VERTEX_BUFFER) + out |= SGX_BIND_VERTEX_BUFFER; + if (bind & PIPE_BIND_INDEX_BUFFER) + out |= SGX_BIND_INDEX_BUFFER; + return out; +} + +static struct pipe_resource * +sgx_pipe_resource_create(struct pipe_screen *ps, + const struct pipe_resource *templ) +{ + struct sgx_pipe_screen *s = (struct sgx_pipe_screen *)ps; + struct sgx_pipe_resource *r = CALLOC_STRUCT(sgx_pipe_resource); + enum sgx_format fmt; + uint32_t w, h; + unsigned bind; + + if (!r) + return NULL; + /* An array target has no descriptor on this part and the screen + * refuses every format for one. Refused here too rather than falling + * through to the 2D path, which would drop array_size and hand the + * caller one layer of the texture it asked for. */ + if (templ->target == PIPE_TEXTURE_1D_ARRAY || + templ->target == PIPE_TEXTURE_2D_ARRAY || + templ->target == PIPE_TEXTURE_CUBE_ARRAY) { + sgx_dbg("resource: an array texture has no descriptor on this " + "part\n"); + FREE(r); + return NULL; + } + r->b.b = *templ; + r->b.b.screen = ps; + r->ws = s->ws; + pipe_reference_init(&r->b.b.reference, 1); + /* Buffers only: the threaded context tracks a buffer's valid range and + * its reallocations, and a texture carries none of that. No CPU + * storage - a buffer here is already a mapping the caller writes + * through, so a second copy would only be another thing to keep. */ + if (templ->target == PIPE_BUFFER) + threaded_resource_init(&r->b.b, false); + + if (templ->target == PIPE_BUFFER) { + fmt = SGX_FMT_NONE; + w = templ->width0; + h = 1; + bind = sgx_bind_of_pipe(templ->bind); + /* A buffer is bytes. Sampling one or rendering into one needs + * a surface descriptor and this driver builds none for a + * buffer, so those binds cannot be served - but they are also + * not what the caller is asking for when it wants sixteen + * bytes of scratch. Refusing the whole create failed the + * allocation instead, which is the X server's "Failed to + * allocate 16 bytes PBO due to GL_OUT_OF_MEMORY": glamor asks + * for 16x1 with render target and sampler view set. Drop the + * binds a buffer cannot have and give it memory. */ + bind &= ~(unsigned)(SGX_BIND_SAMPLER_VIEW | + SGX_BIND_RENDER_TARGET); + /* A bind this driver does not name mapped to no window, the + * create failed, and the caller was handed a resource with no + * buffer behind it - every map on which failed for its whole + * life, silently. A plain CPU-side buffer gets a window. */ + if (!bind) + bind = SGX_BIND_VERTEX_BUFFER; + } else if (util_format_is_depth_or_stencil(templ->format)) { + fmt = SGX_FMT_NONE; + w = templ->width0; + h = templ->height0; + bind = SGX_BIND_DEPTH_STENCIL; + } else { + fmt = sgx_format_of(templ->format); + w = templ->width0; + h = templ->height0; + bind = sgx_bind_of_pipe(templ->bind); + if (fmt == SGX_FMT_NONE || !bind) { + FREE(r); + return NULL; + } + } + { + uint32_t nl = templ->last_level + 1u; + int tw = 0; + + /* A power-of-two chain is twiddled, which is the only layout the + * vendor gives a mipmapped texture on this core - opengles2/ + * texmgmt.c:1823-1827 selects the USIZE/VSIZE encoding for any + * MIPMAP texture, and the level offsets are GetMipMapOffset's. + * SGX_TWIDDLE_MIPS=0 keeps the linear chain, on purpose and only + * to compare: a ten-level twiddled 512x512 was measured to stall + * the render in glmark2's texture-filter=mipmap scene, and the + * vendor's layout is byte for byte the one this driver had, so + * the stall is not the layout and is still open - see + * work/feat-mipmap/README.md. A linear chain is only described + * to the unit under SGX_LINEAR_MIPS, sgx_pipe_set_sampler_views(). */ + if (nl > 1 && templ->target != PIPE_BUFFER && + fmt != SGX_FMT_NONE && sgx_resource_can_twiddle(w, h)) { + const char *e = SGX_ENVS("SGX_TWIDDLE_MIPS"); + + tw = !(e && *e && strtoul(e, NULL, 0) == 0); + } + /* A single level was never twiddled either, so every sample of + * an ordinary texture reads a linear surface - which is the + * layout this part is worst at. The chain case stalls for a + * reason of its own; one level has no chain to get wrong. */ + if (nl == 1 && templ->target != PIPE_BUFFER && + fmt != SGX_FMT_NONE && + sgx_resource_can_twiddle(w, h) && SGX_ENVS("SGX_TWIDDLE_1")) { + tw = 1; + if (SGX_ENVS("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: twiddling %ux%u level " + "texture, bind 0x%x\n", w, h, + templ->bind); + } + /* A texture fewer than three texels wide was twiddled here, + * against a defect that turned out not to exist: the narrowtex + * and luminance cases were uploading packed rows without + * setting GL_UNPACK_ALIGNMENT, so GL read the second row four + * bytes in and the driver returned exactly what was asked for. + * Twiddling made no difference to any of them, measured, and + * FIX_HW_BRN_23054 is on revision 112 rather than this part's + * 121. */ + if (SGX_ENVS("SGX_NO_TWIDDLE")) + tw = 0; + /* A cube is six faces in one object at the stride the texture + * unit assumes, so it has an allocator of its own rather than + * six resources or a taller chain. */ + if (templ->target == PIPE_TEXTURE_CUBE) { + if (sgx_resource_create_cube(&r->r, s->ws, fmt, w, + bind, nl) != + SGX_RESOURCE_OK) { + FREE(r); + return NULL; + } + } else if (templ->target == PIPE_TEXTURE_3D) { + /* Every axis a log2: a volume that is not a power of + * two in all three has no descriptor, and is refused + * here as GL_OUT_OF_MEMORY rather than described at + * the next size up, which would scale its + * coordinates. */ + enum sgx_resource_status st = + sgx_resource_create_3d(&r->r, s->ws, fmt, w, h, + templ->depth0, bind, nl); + + if (st != SGX_RESOURCE_OK) { + sgx_dbg("resource: %ux%ux%u volume refused: " + "%s\n", w, h, templ->depth0, + sgx_resource_status_name(st)); + FREE(r); + return NULL; + } + } else if (nl > 1 || tw || + sgx_format_chunks(templ->format) > 1 || + sgx_format_block(fmt, NULL, NULL, NULL)) { + /* A texel in planes, or a block format - which has + * only the twiddled form, so it is refused when it + * is not a power of two rather than laid out in a + * way the unit cannot read. */ + enum sgx_resource_status st = + sgx_resource_create_planes(&r->r, s->ws, fmt, + w, h, bind, nl, tw, + sgx_format_chunks(templ->format)); + + if (st != SGX_RESOURCE_OK) { + sgx_dbg("resource: %ux%u fmt %d in %u " + "plane(s) refused: %s\n", w, h, + (int)fmt, + sgx_format_chunks(templ->format), + sgx_resource_status_name(st)); + FREE(r); + return NULL; + } + } else { + /* The sample count too: a multisampled depth surface + * holds a value per sample, so one allocated for a + * single sample would be written past. Colour does not + * grow - the pixel back end downscales it on store - + * so this changes nothing for a colour target. */ + enum sgx_resource_status st = + sgx_resource_create_ms(&r->r, s->ws, fmt, w, h, + bind, + templ->nr_samples > 1 ? + templ->nr_samples : 1u); + + if (st != SGX_RESOURCE_OK) { + sgx_dbg("resource: %ux%u fmt %d pipe bind " + "0x%x -> 0x%x refused: %s\n", w, h, + (int)fmt, templ->bind, bind, + sgx_resource_status_name(st)); + FREE(r); + return NULL; + } + } + } + /* Alpha is not in the caller's format, so it must not come out of the + * surface: an RGB texture sampled with a zero alpha byte disappears + * wherever it is blended by source alpha, which is how ioquake3 draws + * its cinematics - the video played as a black rectangle. */ + r->no_alpha = !util_format_has_alpha(templ->format); + r->zfmt = util_format_is_depth_or_stencil(templ->format) ? + templ->format : PIPE_FORMAT_NONE; + if (SGX_ENVS("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: resource: pipe format %u (%s), alpha %d, " + "%ux%u bind 0x%x\n", templ->format, + util_format_short_name(templ->format), + !r->no_alpha, templ->width0, templ->height0, + templ->bind); + return &r->b.b; +} + +/* Exporting a resource is what lets anything outside this driver see it: gbm + * hands the handle to drmModeAddFB for scanout, and EGL hands the file + * descriptor to another process. Without it a display server gets a null + * pointer where the stride should be, which is where glamor used to stop. */ +static bool sgx_pipe_resource_get_handle(struct pipe_screen *ps, + struct pipe_context *pc, + struct pipe_resource *pres, + struct winsys_handle *whandle, + unsigned usage) +{ + struct sgx_pipe_screen *s = (struct sgx_pipe_screen *)ps; + struct sgx_pipe_resource *r = (struct sgx_pipe_resource *)pres; + int fd; + + (void)pc; (void)usage; + if (!r || !r->r.bo.handle) + return false; + /* The buffer is about to be named outside this driver, where nothing + * of ours will flush the job that still owes it pixels. */ + sgx_twod_flush(s->ws); + /* One plane, one layer. A request for anything else is not something + * this driver can answer, and answering it with plane zero would be + * worse than refusing. */ + if (whandle->plane || whandle->layer) + return false; + /* The buffer is about to be someone else's - a compositor's texture, + * a scanout - and what is in it has to be the frame that was drawn, + * not the one before. */ + sgx_wait_idle(s->ws); + + whandle->stride = r->r.stride; + whandle->offset = 0; + whandle->modifier = DRM_FORMAT_MOD_LINEAR; + + switch (whandle->type) { + case WINSYS_HANDLE_TYPE_KMS: + case WINSYS_HANDLE_TYPE_SHARED: + whandle->handle = r->r.bo.handle; + return true; + case WINSYS_HANDLE_TYPE_FD: + fd = sgx_winsys_fd(s->ws); + if (fd < 0) + return false; + if (drmPrimeHandleToFD(fd, r->r.bo.handle, DRM_CLOEXEC, + (int *)&whandle->handle)) + return false; + return true; + default: + return false; + } +} + +/* The other half of resource_get_handle: taking an object somebody else + * allocated and giving it a pipe_resource. A display server hands its scanout + * buffer to GL this way - gbm allocates it, EGL imports it, and glamor draws + * into the result. Without this the import fails, glamor gets no texture for + * the screen, and nothing it draws ever reaches the display. */ +static struct pipe_resource * +sgx_pipe_resource_from_handle(struct pipe_screen *ps, + const struct pipe_resource *templ, + struct winsys_handle *whandle, + unsigned usage) +{ + struct sgx_pipe_screen *s = (struct sgx_pipe_screen *)ps; + struct sgx_pipe_resource *r; + enum sgx_format fmt; + uint32_t gem = 0, want; + unsigned bpp; + int fd, own = 0; + + (void)usage; + if (!templ || !whandle || templ->target == PIPE_BUFFER) + return NULL; + fmt = sgx_format_of(templ->format); + if (fmt == SGX_FMT_NONE) + return NULL; + bpp = xpsb_format_bpp((uint32_t)fmt); + if (!bpp) + return NULL; + /* There is no pitch field in the descriptor, so an imported surface + * whose rows are not where the hardware will look for them cannot be + * accepted - it would be sampled and rendered shifted. */ + want = bpp * ((templ->width0 + 31u) & ~31u); + if (whandle->stride != want) { + sgx_dbg("import: stride %u is not the %u this hardware " + "reads\n", whandle->stride, want); + return NULL; + } + + fd = sgx_winsys_fd(s->ws); + switch (whandle->type) { + case WINSYS_HANDLE_TYPE_KMS: + case WINSYS_HANDLE_TYPE_SHARED: + gem = whandle->handle; + break; + case WINSYS_HANDLE_TYPE_FD: + if (fd < 0 || + drmPrimeFDToHandle(fd, (int)whandle->handle, &gem)) + return NULL; + own = 1; + break; + default: + return NULL; + } + + r = CALLOC_STRUCT(sgx_pipe_resource); + if (!r) + return NULL; + r->b.b = *templ; + r->b.b.screen = ps; + r->ws = s->ws; + pipe_reference_init(&r->b.b.reference, 1); + if (templ->target == PIPE_BUFFER) + threaded_resource_init(&r->b.b, false); + r->r.format = fmt; + r->r.width = templ->width0; + r->r.height = templ->height0; + r->r.stride = whandle->stride; + r->r.bind = sgx_bind_of_pipe(templ->bind) | SGX_BIND_RENDER_TARGET; + /* The raster pass writes past the last row, as it does for a target + * this driver allocated itself. */ + r->r.size = (uint32_t)(((uint64_t)whandle->stride * + (templ->height0 + 32u) + 0xfffu) & ~0xfffull); + r->r.bo.handle = gem; + r->r.bo.size = r->r.size; + /* A handle that arrived by name rather than by descriptor belongs to + * its maker; closing it here would pull it out from under them. */ + r->r.foreign = !own; + if (sgx_bo_bind(s->ws, &r->r.bo, + (uint32_t)sgx_resource_window(r->r.bind))) { + sgx_dbg("import: %ux%u would not bind\n", r->r.width, + r->r.height); + FREE(r); + return NULL; + } + r->r.bound = 1; + sgx_dbg("import: %ux%u stride %u handle %u -> 0x%08x\n", r->r.width, + r->r.height, r->r.stride, gem, (uint32_t)r->r.bo.gpu_va); + return &r->b.b; +} + +/* Give dst src's storage and take src's away. The threaded context queues + * buffer invalidations, so it allocates a fresh buffer, replays the writes + * into it and then asks for the two to be swapped - which is why a mapping + * taken before the swap must not be written through afterwards. + * + * Nothing is rebound. A binding holds the pipe_resource, and that object is + * unchanged here; only the storage behind it moves, and every path in this + * driver reads sgx_resource out of the resource at the point it uses it. */ +static void sgx_pipe_replace_buffer_storage(struct pipe_context *pc, + struct pipe_resource *dst, + struct pipe_resource *src, + unsigned num_rebinds, + uint32_t rebind_mask, + uint32_t delete_buffer_id) +{ + struct sgx_pipe_resource *d = (struct sgx_pipe_resource *)dst; + struct sgx_pipe_resource *o = (struct sgx_pipe_resource *)src; + struct sgx_resource tmp; + + (void)pc; (void)num_rebinds; (void)rebind_mask; (void)delete_buffer_id; + if (!d || !o) + return; + /* The part may still be reading what dst names. */ + sgx_wait_idle(d->ws); + tmp = d->r; + d->r = o->r; + o->r = tmp; + d->write_mapped = 0; + o->write_mapped = 0; +} + +static void sgx_pipe_resource_destroy(struct pipe_screen *ps, + struct pipe_resource *pres) +{ + struct sgx_pipe_resource *r = (struct sgx_pipe_resource *)pres; + + (void)ps; + if (!r) + return; + if (pres->target == PIPE_BUFFER) + threaded_resource_deinit(pres); + /* Before anything is unbound or freed: the render on the core names + * this buffer by address, and taking the address away underneath it + * faults rather than skips. */ + sgx_wait_idle(r->ws); + /* A resource still bound somewhere must be taken out of it first. A + * pixmap glamor drops while it is the frame's texture left the unit's + * fixed address unmapped, and the frame samples that address whatever + * the program does - so every later render faulted there and X could + * no longer repaint. Measured after a GL client exits: the screen kept + * its last frame and xsetroot changed nothing. */ + { + struct sgx_pipe_screen *s = (struct sgx_pipe_screen *)ps; + struct sgx_pipe_context *c; + + for (c = s ? s->ctxs : NULL; c; c = c->next) { + unsigned u; + + /* Send the frame first if one of its records names + * this buffer: the record carries the address, not a + * reference, so the buffer cannot go until the frame + * that names it has. */ + if (sgx_context_frame_uses(&c->ctx, &r->r.bo)) { + unsigned nd_destroy = c->ctx.draws; + + sgx_dbg("resource destroy: the frame still " + "names this texture, sending it\n"); + sgx_perf_note("buffer still named"); + sgx_flush(&c->ctx); + if (nd_destroy) + sgx_set_clear_enable(&c->ctx, 0); + } + for (u = 0; u < 2; u++) + if (c->ctx.tex_bo[u] == &r->r.bo) + sgx_release_texture(&c->ctx, u); + sgx_frame_drop_tex(&c->ctx, &r->r.bo); + if (c->ctx.target_bo == &r->r.bo) { + c->ctx.target_bo = NULL; + c->ctx.target_prev_va = 0; + c->ctx.cookie_valid = 0; + } + if (c->cur_tex == pres) + c->cur_tex = NULL; + if (c->frame_tex == pres) + c->frame_tex = NULL; + } + } + sgx_resource_destroy(&r->r, r->ws); + free(r->zstage); + free(r->zrow); + FREE(r); +} + +/* --- fences ------------------------------------------------------------- */ + +static struct sgx_pipe_fence sgx_the_fence; + +static void sgx_pipe_fence_reference(struct pipe_screen *ps, + struct pipe_fence_handle **dst, + struct pipe_fence_handle *src) +{ + (void)ps; + *dst = src; +} + +static bool sgx_pipe_fence_finish(struct pipe_screen *ps, + struct pipe_context *pc, + struct pipe_fence_handle *fence, + uint64_t timeout) +{ + struct sgx_pipe_screen *s = (struct sgx_pipe_screen *)ps; + + (void)pc; (void)fence; (void)timeout; + /* There is one fence and it stands for "everything submitted so far", + * because the driver submits a whole frame at a time. What it costs is + * the render the last submit left running: this is where glFinish and + * the frontend's end-of-frame synchronisation land, so it is the last + * point before a presenting compositor reads the pixels. */ + /* The result matters: returning true after a failed wait tells the + * frontend the frame arrived, and a locked core then looks like a + * healthy one that is merely slow. */ + if (s && sgx_wait_idle(s->ws)) + return false; + return true; +} + +/* GL asks this after a wait fails, and it is how an application learns its + * context lost work rather than guessing from a frozen picture. */ +/* Once per loss, which is what GL asks for: glGetGraphicsResetStatus reports + * the reset and then returns NO_ERROR again. Reporting the sticky flag would + * answer "reset" for the life of the context after one bad frame - and this + * driver recovers the core, so a stalled frame is not a device that has gone + * away. A context that has never seen a loss starts level with the winsys. */ +static enum pipe_reset_status sgx_pipe_get_device_reset_status( + struct pipe_context *pc) +{ + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + unsigned gen; + + if (!c || !c->ctx.ws) + return PIPE_NO_RESET; + gen = sgx_winsys_lost_gen(c->ctx.ws); + if (gen == c->reset_gen) + return PIPE_NO_RESET; + c->reset_gen = gen; + return PIPE_UNKNOWN_CONTEXT_RESET; +} + +static int sgx_pipe_get_screen_fd(struct pipe_screen *ps) +{ + return ((struct sgx_pipe_screen *)ps)->fd; +} + +struct pipe_screen *sgx_screen_create(int fd); +struct pipe_screen *sgx_screen_create_ws(struct sgx_winsys *ws); +struct pipe_context *sgx_context_create(struct pipe_screen *screen, + void *priv, unsigned flags); + +/* pipe_context::destroy. Mesa calls this unconditionally when a context goes + * away - it is not optional, and leaving it NULL is a crash on teardown rather + * than a missing feature. */ +static void sgx_pipe_context_destroy(struct pipe_context *pc) +{ + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + + if (!c) + return; + if (c->screen) { + struct sgx_pipe_context **pp = &c->screen->ctxs; + + while (*pp && *pp != c) + pp = &(*pp)->next; + if (*pp) + *pp = c->next; + } + /* Nothing the context owns may go while the core is still reading it, + * and everything a frame is built from is the context's. */ + sgx_wait_idle(c->ctx.ws); + free(c->hwidx); + free(c->hwmap); + free(c->hwsrc); + if (c->base.stream_uploader) + u_upload_destroy(c->base.stream_uploader); + if (c->draw) + draw_destroy(c->draw); + /* c->vbuf is not released here: draw_set_render() hands it to the draw + * module, and draw_destroy() above destroys it. Freeing it again is a + * double free that aborts every glmark2-drm run at teardown. */ + if (c->blitter) { + util_blitter_destroy(c->blitter); + c->blitter = NULL; + } + util_unreference_framebuffer_state(&c->fb); + free(c->hwrec); + c->hwrec = NULL; + c->hwrec_floats = 0; + c->hwidx = NULL; + c->hwidx_n = 0; + c->hwmap = NULL; + c->hwsrc = NULL; + c->hwmap_n = 0; + c->hwsrc_n = 0; + /* Comfortably more than a draw this driver sees - ioquake3's are a few + * hundred corners - and small enough that the block always fits. */ + c->hwidx_max_n = 0; + free(c->hwel); + c->hwel = NULL; + c->hwel_n = 0; + free(c->swz); + c->swz = NULL; + c->nswz = 0; + sgx_dbg("context destroy\n"); + /* releases the objects the context owns */ + sgx_context_fini(&c->ctx); + free(c); +} + +/* pipe_screen::destroy, likewise. The screen owns the winsys it created from + * the file descriptor, so it closes it. */ +static void sgx_pipe_screen_destroy(struct pipe_screen *ps) +{ + struct sgx_pipe_screen *s = (struct sgx_pipe_screen *)ps; + + if (!s) + return; + slab_destroy_parent(&s->transfer_pool); + if (s->ws) + sgx_winsys_destroy(s->ws); + free(s); +} + +/* Acquiring the device and building the screen are separate, because they fail + * for different reasons and because a test needs the second without the first. + * sgx_winsys_create_drm() checks the device really is one of ours - it reads + * the address windows back through GET_PARAM - so an fd for something else + * fails here rather than at the first submit. */ +struct pipe_screen *sgx_screen_create(int fd) +{ + struct sgx_winsys *ws = NULL; + struct pipe_screen *ps; + + if (fd >= 0) { + ws = sgx_winsys_create_drm(fd); + if (!ws) { + sgx_dbg("screen: no winsys for fd %d\n", fd); + return NULL; + } + } + ps = sgx_screen_create_ws(ws); + /* What binary this is. Four fixes were once measured as having no + * effect at all, and the logs of the two runs turned out identical + * but for heap addresses - the driver under test had not been + * replaced. A log that names its own build cannot hide that, and + * the target carries more than one copy of this library + * (/root/sgxmesa/lib against /usr/lib). */ + sgx_dbg("screen: fd %d -> %s (build " __DATE__ " " __TIME__ ")\n", + fd, ps ? "created" : "NULL"); + return ps; +} + +struct pipe_screen *sgx_screen_create_ws(struct sgx_winsys *ws) +{ + struct sgx_pipe_screen *s = calloc(1, sizeof(*s)); + + if (!s) + return NULL; + s->fd = ws ? sgx_winsys_fd(ws) : -1; + s->ws = ws; + slab_create_parent(&s->transfer_pool, sizeof(struct sgx_transfer), 16); + s->base.destroy = sgx_pipe_screen_destroy; + s->base.get_name = sgx_pipe_get_name; + s->base.get_vendor = sgx_pipe_get_vendor; + s->base.get_device_vendor = sgx_pipe_get_device_vendor; + s->base.is_format_supported = sgx_pipe_is_format_supported; + s->base.context_create = sgx_context_create; + s->base.resource_create = sgx_pipe_resource_create; + s->base.resource_get_handle = sgx_pipe_resource_get_handle; + s->base.resource_from_handle = sgx_pipe_resource_from_handle; + s->base.resource_destroy = sgx_pipe_resource_destroy; + s->base.fence_reference = sgx_pipe_fence_reference; + s->base.fence_finish = sgx_pipe_fence_finish; + s->base.get_screen_fd = sgx_pipe_get_screen_fd; + s->base.nir_options[MESA_SHADER_VERTEX] = &sgx_vs_nir_options; + s->base.nir_options[MESA_SHADER_FRAGMENT] = &sgx_fs_nir_options; + sgx_pipe_init_caps(&s->base); + sgx_pipe_init_shader_caps(&s->base); + return &s->base; +} + +/* --- pipe_context ------------------------------------------------------ + * + * State objects are created, bound and deleted separately in Gallium, and the + * translation belongs at *create* rather than at bind: Gallium creates a state + * object once and binds it many times, so converting on every bind would + * repeat work the API is shaped to avoid. + */ + +static void *sgx_pipe_create_dsa(struct pipe_context *pc, + const struct pipe_depth_stencil_alpha_state *in) +{ + struct sgx_dsa_state *d = calloc(1, sizeof(*d)); + unsigned i; + + (void)pc; + if (!d) + return NULL; + /* A knob to take the depth test out of the picture while a rendering + * defect is being narrowed down; not a supported mode. */ + d->depth_enabled = SGX_ENVS("SGX_NO_DEPTH") ? 0 : in->depth_enabled; + d->depth_writemask = in->depth_writemask; + d->depth_func = sgx_func_of(in->depth_func); + for (i = 0; i < 2; i++) { + d->stencil[i].enabled = in->stencil[i].enabled; + d->stencil[i].func = sgx_func_of(in->stencil[i].func); + d->stencil[i].fail_op = sgx_stencil_op_of(in->stencil[i].fail_op); + d->stencil[i].zfail_op = sgx_stencil_op_of(in->stencil[i].zfail_op); + d->stencil[i].zpass_op = sgx_stencil_op_of(in->stencil[i].zpass_op); + d->stencil[i].valuemask = in->stencil[i].valuemask; + d->stencil[i].writemask = in->stencil[i].writemask; + } + return d; +} + +/* The bound state with the reference values folded in. Gallium sets the + * references through set_stencil_ref rather than in the state object, and + * the ISP takes them from word A bits [7:0] - so a state bound with the + * object's own zero compared every fragment against a reference of 0. */ +static void sgx_pipe_apply_dsa(struct sgx_pipe_context *c) +{ + struct sgx_dsa_state d; + + if (!c->cur_dsa) + return; + d = *(const struct sgx_dsa_state *)c->cur_dsa; + d.stencil_ref = c->stencil_ref.ref_value[0]; + d.stencil_ref_back = c->stencil_ref.ref_value[1]; + sgx_bind_dsa(&c->ctx, &d); +} + +static void sgx_pipe_bind_dsa(struct pipe_context *pc, void *state) +{ + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + + c->cur_dsa = state; + sgx_pipe_apply_dsa(c); +} + +static void sgx_pipe_delete_state(struct pipe_context *pc, void *state) +{ + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + unsigned i; + + /* Forget it first. These are tracked only so the blitter can put the + * caller's state back, and the blitter binds what it is handed: an + * entry left pointing at a deleted object is bound after it has been + * freed, which aborted the server inside malloc rather than anywhere + * near here. */ + if (c) { + if (c->cur_dsa == state) + c->cur_dsa = NULL; + if (c->cur_rast == state) + c->cur_rast = NULL; + if (c->cur_blend == state) + c->cur_blend = NULL; + for (i = 0; i < SGX_MAX_TEX_UNITS; i++) + if (c->cur_samplers[i] == state) + c->cur_samplers[i] = NULL; + } + free(state); +} + +/* Fragment shaders need more than a free(): the context, the frame and every + * draw record that has been opened this frame hold raw pointers to the object, + * and the upload path reads ->compiled, ->ninsns and ->code[] out of whatever + * they point at. Freeing one while a record still names it uploads freed + * memory to the USSE, and a program without .end runs on. The record pointers + * are compared as addresses too, so a re-used allocation would compare equal + * and suppress the split that gives a record its own state. */ +static void sgx_pipe_delete_fs(struct pipe_context *pc, void *state) +{ + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + + if (!state) + return; + if (c->cur_fs == state) + c->cur_fs = NULL; + if (c->frame_fs == state) + c->frame_fs = NULL; + sgx_forget_fs(&c->ctx, state); + sgx_shader_fini((struct sgx_shader *)state); + /* Kept past the compiled form so the program could be built again; + * sgx_shader_fini() carries it, so it is released here. */ + free(((struct sgx_shader *)state)->src_insns); + free(((struct sgx_shader *)state)->src_io); + free(((struct sgx_shader *)state)->swz_key); + free(((struct sgx_shader *)state)->cls_key); + free(((struct sgx_shader *)state)->cmp_key); + ralloc_free(((struct sgx_shader *)state)->src_nir); + free(state); +} + +static void *sgx_pipe_create_rast(struct pipe_context *pc, + const struct pipe_rasterizer_state *in) +{ + struct sgx_pipe_rast *r = calloc(1, sizeof(*r)); + + (void)pc; + if (!r) + return NULL; + /* Gallium's copy is kept as well as the translation: the draw module + * is handed the state it was given, not the hardware's view of it. */ + r->pipe = *in; + /* Except this: the wide-point stage reads the fragment shader out of + * the draw module to find a sprite coordinate, and this driver never + * registers one, so it asserted. Sprites are handled where points are + * widened, in sgx_vbuf_widen_points(): the vbuf is told which + * coordinates to write and the draw module is shown none, so its own + * stage - which would take every point away from the vbuf - stays + * out of the pipeline (draw_pipe_validate.c: sprite_coord_enable and + * point_quad_rasterization both put it in). */ + r->sprite_enable = in->sprite_coord_enable; + r->sprite_upper_left = + in->sprite_coord_mode == PIPE_SPRITE_COORD_UPPER_LEFT; + r->pipe.point_quad_rasterization = 0; + r->pipe.sprite_coord_enable = 0; + /* And this: a per-vertex size puts the draw module's wide-point stage + * in the pipeline whatever the threshold says, and that stage sizes + * its quad against the viewport - which this driver's epilogue has + * already applied, so a one-pixel point came out twenty-three wide. + * Points are widened in sgx_vbuf_widen_points() instead, at the size + * below, in the window coordinates the vertex is already in. */ + /* The draw module is the only thing that computes gl_PointSize, and + * with this clear that output is never written - so clearing it threw + * every per-vertex point size away and drew them all at the + * rasterizer's constant. It was cleared because the module's own + * wide-point stage sizes against a viewport the vertex epilogue has + * already applied; that stage does not run here, and the size is read + * out of the record by sgx_vbuf_widen_points() instead. + * SGX_NO_PT_PER_VERTEX goes back to the constant. */ + r->pipe.point_size_per_vertex = SGX_ENVS("SGX_NO_PT_PER_VERTEX") ? 0 : + in->point_size_per_vertex; + /* Gallium's cull_face is a bitmask of faces; the hardware has one + * three-valued field, so culling both is not expressible and is + * reported as culling the back rather than silently culling one. */ + r->hw.cull_face = (in->cull_face == PIPE_FACE_FRONT) ? 1u : + (in->cull_face == PIPE_FACE_NONE) ? 0u : 2u; + r->hw.front_ccw = in->front_ccw; + r->hw.flatshade = in->flatshade; + r->hw.flatshade_first = in->flatshade_first; + /* The ISP's depth bias, applied only under the part's transform: on + * the draw module's path the module adds it to the vertices itself. */ + r->hw.offset_tri = in->offset_tri; + r->hw.offset_units = in->offset_units; + r->hw.offset_scale = in->offset_scale; + r->scissor = in->scissor; + /* The flip predicate, which decides which device-space winding the + * part throws away. gl-re/state-model.md:852 ties it to whether the + * viewport transform flips y, and this frame's always does - but with + * it clear, culling removed every triangle rather than half of them, + * so the sense is being established on hardware rather than assumed. */ + r->hw.target_y_flipped = SGX_ENVS("SGX_YFLIP") ? 1 : 0; + /* Set when the draw actually runs on the part; the rasterizer state is + * bound before that is known, so the context corrects it per draw. */ + r->hw.hw_transform = 0; + return r; +} + +/* The corner a flat-shaded colour comes from, for the iterator that carries + * it. The MTE's shade model is not enough on its own: the vendor writes the + * iterator's own DOUTI FLATSHADE field from the same model, on whichever + * issue delivers a colour, and gouraud there is a written value rather than + * the reset one (opengles1/usegles.c:2240-2252). That holds for the base + * colour on USEISSUE_V0 as much as for a colour routed through a coordinate + * set under FIX_HW_BRN_25211 - the vendor's test is on whether the issue + * samples, not on which slot the colour arrived in - so both shapes settle + * here. The decomposition already puts GL's provoking vertex at the corner + * sgx_raster_word() names, so every unit names the same one. */ +static void sgx_pipe_settle_flatshade(struct sgx_pipe_context *c) +{ + struct sgx_shader *fs = (struct sgx_shader *)c->cur_fs; + + if (!fs || !(fs->attribs.colour_set || fs->attribs.colour)) + return; + fs->attribs.flatshade = c->ctx.rast.flatshade ? + (c->ctx.rast.flatshade_first ? 1u : 3u) : 0u; +} + +static void sgx_pipe_bind_rast(struct pipe_context *pc, void *state) +{ + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + struct sgx_pipe_rast *r = state; + + c->cur_rast = state; + + /* Before the driver's own, as the draw module's note asks: it binds + * its own rasterizer state and would undo this. */ + c->ctx.line_width = r ? r->pipe.line_width : 1.0f; + c->ctx.point_size = r ? r->pipe.point_size : 1.0f; + sgx_vbuf_set_sprite(c->vbuf, r ? r->sprite_enable : 0u, + r ? r->sprite_upper_left : 1); + sgx_vbuf_set_psize_per_vertex(c->vbuf, + r ? r->pipe.point_size_per_vertex : 1); + draw_set_rasterizer_state(c->draw, r ? &r->pipe : NULL, state); + /* Both, or neither. Mesa unbinds by binding NULL - cso_unbind_context() + * does it for every state object when a context is destroyed - so the + * scissor line running unconditionally dereferenced NULL and took + * glmark2 down on teardown. */ + if (r) { + sgx_bind_rasterizer(&c->ctx, &r->hw); + c->ctx.scissor_enable = r->scissor; + } else { + c->ctx.scissor_enable = 0; + } + /* The shade model reaches the iterator as well as the MTE once a + * colour rides in a coordinate set, so this bind has to reach the + * fragment program's attributes. */ + sgx_pipe_settle_flatshade(c); +} + +static void sgx_pipe_set_framebuffer(struct pipe_context *pc, + const struct pipe_framebuffer_state *fb) +{ + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + struct sgx_framebuffer f; + unsigned cb; + + /* Anything about to be rendered into cannot keep a staging copy taken + * before the render: the part writes the surface and the copy would + * still hold what was there. */ + for (cb = 0; fb && cb < fb->nr_cbufs; cb++) + if (fb->cbufs[cb].texture) + ((struct sgx_pipe_resource *) + fb->cbufs[cb].texture)->r.stage_valid = 0; + if (fb && fb->zsbuf.texture) + ((struct sgx_pipe_resource *)fb->zsbuf.texture)->r.stage_valid = 0; + + /* A frame renders into one target. Draws already accumulated belong to + * the old one, so they have to go before the new one takes its place - + * otherwise they land in whatever is bound when the frame is finally + * submitted. A display server hits this the moment it fills a pixmap + * and copies it to a window. */ + if (c->ctx.draws || sgx_clear_pending(&c->ctx)) { + unsigned n = c->ctx.draws; + sgx_perf_note("render target"); + int ret = sgx_flush(&c->ctx); + + sgx_dbg("framebuffer: flushed %u draw(s) for the old target: " + "%d\n", n, ret); + /* This flush is the context's, not the pipe's, so the dump in + * sgx_pipe_flush() never sees these frames - which is every + * frame a display server renders into a pixmap. */ + { + const char *path = SGX_ENVS("SGX_DUMP_RT"); + const char *wo = SGX_ENVS("SGX_DUMP_RT_W"); + + if (path && wo && *wo && !ret && + c->ctx.fb.width && + c->ctx.fb.width <= (unsigned)atoi(wo)) { + static unsigned seq; + char nm[256]; + + snprintf(nm, sizeof nm, "%s-sw%u.ppm", path, + seq++); + sgx_dump_target(&c->ctx, nm); + } + if (path && wo) + sgx_dbg("framebuffer: at switch the target is " + "%ux%u\n", c->ctx.fb.width, + c->ctx.fb.height); + } + } + memset(&f, 0, sizeof(f)); + f.cbuf_format = SGX_CBUF_FMT_NONE; + f.width = fb->width; + f.height = fb->height; + f.has_zsbuf = fb->zsbuf.texture != NULL; + if (fb->nr_cbufs) + f.cbuf_format = + (uint32_t)sgx_format_of(fb->cbufs[0].format); + /* The sample count the frame is built at. util_framebuffer_get_num_ + * samples() is the same rule mesa/st uses to decide whether the sample + * mask is live: a pipe_surface's own nr_samples when it has one - the + * EXT_multisampled_render_to_texture case, a multisampled render into + * single-sampled memory, which is what this part actually does - and + * the resource's otherwise. + * + * A count the part cannot reach is not silently rounded: it is left + * at one and said, because a frame drawn at a sample count the scene + * was not built for is a picture with no explanation. */ + f.samples = util_framebuffer_get_num_samples(fb); + if (!xpsb_msaa_axis(f.samples)) { + fprintf(sgx_log(), "sgx: %u samples a pixel is not a mode this " + "part has (one or 2x2); the framebuffer is built " + "single sampled\n", f.samples); + f.samples = 1; + } + c->fb_samples = f.samples; + + util_copy_framebuffer_state(&c->fb, fb); + /* The draw module's polygon offset stage scales glPolygonOffset's + * units by the depth format's minimum resolvable difference, which + * it takes from here - and starts at zero, so without this every + * offset it applied was nothing. */ + if (fb->zsbuf.texture) + draw_set_zs_format(c->draw, fb->zsbuf.format); + /* The hardware renders into the caller's colour buffer, so the object + * behind it takes the target's address before the frame around it is + * built. */ + if (fb->nr_cbufs && fb->cbufs[0].texture) { + struct sgx_pipe_resource *r = + (struct sgx_pipe_resource *)fb->cbufs[0].texture; + int ret; + + /* Settled here rather than at creation: Mesa gives every + * colour-renderable texture the render-target bind whether or + * not the caller ever attaches it, so a mipmapped texture and + * an FBO attachment are indistinguishable until this. The + * raster pass writes a linear surface, so one that is really + * attached cannot stay twiddled. */ + if (r->r.twiddled) { + enum sgx_resource_status rs = + sgx_resource_make_linear(&r->r, c->ctx.ws); + + sgx_dbg("framebuffer: %ux%u relaid out linearly to be " + "rendered into: %s\n", r->r.width, r->r.height, + sgx_resource_status_name(rs)); + r->r.stage_valid = 0; + } + ret = sgx_use_target(&c->ctx, &r->r.bo); + + if (ret) { + sgx_dbg("framebuffer: the colour buffer would not take " + "the target address: %d\n", ret); + return; + } + } else { + /* No colour attachment: the caller's last colour buffer must + * not stay the target. The part writes colour whatever the + * framebuffer declares, and a depth-only pass is routinely + * larger than the window that preceded it - glmark2's shadow + * map is 2048x1151 over a 1280x720 window - so the pixel back + * end wrote past the end of it and faulted. Released here, the + * frame is given one of the context's own at the size it + * renders. */ + /* Only for a framebuffer that will be set up. Gallium also + * unbinds by setting a 0x0 one, and sgx_set_framebuffer() + * refuses that before it reallocates - so releasing there left + * the target address mapped by nothing and the next render + * faulted at 0x80000000. */ + if (fb->width && fb->height) + sgx_release_target(&c->ctx); + } + /* And the depth attachment, for the same reason: rendering into a + * depth texture is what a shadow map is, and the frame writes the + * object whose address sits in the depth slot. The size the raster + * pass writes is the padded one - the stride rounded to 32 and 32 + * rows past the height, four bytes a pixel - which is what + * sgx_resource_create() gives a depth resource, so a matching + * attachment fits exactly. One that does not is left alone and the + * frame keeps its own buffer; the caller then samples an unwritten + * texture, so it is said rather than left to the picture. */ + if (fb->zsbuf.texture) { + struct sgx_pipe_resource *z = + (struct sgx_pipe_resource *)fb->zsbuf.texture; + /* At the sample resolution: depth is held per sample and the + * frame stores it that way, so an attachment sized for one + * sample does not hold a 2x2 frame's. */ + uint64_t need = sgx_resource_depth_bytes(fb->width, fb->height, + f.samples); + int ret = z->r.bo.handle ? + sgx_use_depth(&c->ctx, &z->r.bo, need) : -EINVAL; + + if (ret) { + fprintf(sgx_log(), "sgx: the depth attachment (%ux%u, " + "%llu byte(s), needs %llu) will not hold the " + "frame's depth: %d - it is rendered into the " + "context's own buffer and reads unwritten\n", + z->r.width, z->r.height, + (unsigned long long)z->r.bo.size, + (unsigned long long)need, ret); + sgx_release_depth(&c->ctx); + } + } else { + sgx_release_depth(&c->ctx); + } + { + int ret = sgx_set_framebuffer(&c->ctx, &f); + + sgx_dbg("framebuffer: target bo %u gpu_va 0x%llx\n", + c->ctx.bo[SGX_CTX_BO_TARGET].handle, + (unsigned long long) + c->ctx.bo[SGX_CTX_BO_TARGET].gpu_va); + sgx_dbg("framebuffer: %ux%u, %u colour buffer(s), depth %d: %d\n", + f.width, f.height, fb->nr_cbufs, f.has_zsbuf, ret); + if (fb->nr_cbufs && fb->cbufs[0].texture) + sgx_dbg(" target format %u, hw code %d\n", + (unsigned)fb->cbufs[0].format, + (int)sgx_format_of(fb->cbufs[0].format)); + } +} + +/* State the driver cannot honour. + * + * Unconditional: the consequence is a wrong picture, and behind SGX_DEBUG that + * was invisible - a caller saw unblended or unmasked output with nothing to + * say why. Reported once per site, keyed on the literal, because these sit on + * the state-change path and would otherwise repeat per draw. */ +static void sgx_unsupported(const char *what) +{ + static const char *said[16]; + static unsigned n; + unsigned i; + + for (i = 0; i < n; i++) + if (said[i] == what) + return; + if (n < sizeof said / sizeof said[0]) + said[n++] = what; + fprintf(sgx_log(), "sgx: %s - what is drawn will not match what was " + "asked for\n", what); +} + +/* pipe_context::clear. The frame clears as it begins - the clearing draws are + * part of it - so this is the colour they write rather than a separate pass. + * Depth and stencil are cleared by the same frame whenever depth is on, which + * is why only the colour is taken here. */ +static void sgx_blitter_save(struct sgx_pipe_context *c); +static void sgx_pipe_submit_pending_why(struct sgx_pipe_context *c, + const char *why); + +static void sgx_pipe_clear(struct pipe_context *pc, unsigned buffers, + uint32_t color_mask, uint8_t stencil_mask, + const struct pipe_scissor_state *scissor, + const union pipe_color_union *color, + double depth, unsigned stencil) +{ + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + uint32_t packed; + + (void)scissor; + /* color_mask is a per-channel write mask, so all four channels is 0xf; + * stencil_mask is the bits of the stencil buffer to clear. Neither is + * honoured here - the frame clears as it begins, with one colour and + * no mask - so say so where the caller asked for either. */ + if ((buffers & PIPE_CLEAR_COLOR) && (color_mask & 0xfu) != 0xfu) + sgx_unsupported("a colour clear with a write mask clears " + "every channel"); + sgx_dbg("clear: buffers 0x%x, colour %s, depth %g, stencil %u\n", + buffers, color ? "given" : "none", depth, stencil); + /* Depth is its own request: the two used to share one flag, so a + * glClear(GL_DEPTH_BUFFER_BIT) with no colour returned here and the + * depth buffer kept the previous frame's values. */ + if (buffers & PIPE_CLEAR_DEPTH) { + sgx_set_clear_depth(&c->ctx, (float)depth); + sgx_set_zclear_enable(&c->ctx, 1); + } + /* The stencil shares the depth surface and is cleared with it: a + * frame that does not load its depth starts every tile from the + * background object, whose stencil this sets. Clearing the stencil + * while keeping the depth is a clearing object, as the vendor draws + * it (opengles1/clear.c:884-895): a full-target quad that writes no + * colour and no depth, with SREF and a word C of ALWAYS/REPLACE. + * The blitter's stencil-only clear binds exactly that state - a + * zero colour mask, which is TAGWRITEDIS here, and a keep-depth, + * write-stencil DSA - so the clear is a draw through the ordinary + * path. When a clear is already pending and nothing has been drawn + * since, the background object does it and only the value is set. */ + if (buffers & PIPE_CLEAR_STENCIL) { + if ((stencil_mask & 0xffu) != 0xffu) + sgx_unsupported("a stencil clear with a write mask " + "clears every bit"); + if ((buffers & PIPE_CLEAR_DEPTH) || + (c->ctx.clear_pending && !c->ctx.draws)) { + sgx_set_clear_stencil(&c->ctx, stencil); + } else if (c->blitter && c->fb.zsbuf.texture) { + sgx_dbg("clear: stencil %u by a clearing object, the " + "depth kept\n", stencil); + sgx_blitter_save(c); + util_blitter_clear(c->blitter, c->fb.width, + c->fb.height, 1, PIPE_CLEAR_STENCIL, + NULL, 0.0, stencil, false); + buffers &= ~PIPE_CLEAR_STENCIL; + } else { + sgx_unsupported("a stencil clear that keeps the depth " + "is not performed without a blitter"); + } + } + if (!(buffers & PIPE_CLEAR_COLOR) || !color) { + if (buffers & PIPE_CLEAR_DEPTH) + sgx_request_clear(&c->ctx); + return; + } + /* Red high, blue low - the same order the fragment path packs in. + * + * This was the other way round for a while, on the strength of a blue + * clear looking red on the panel. That observation came through the + * framebuffer blit, which has a channel order of its own; read back + * through glReadPixels, which is defined by GL rather than by how a + * tool happens to interpret the bytes, the reversed order put red and + * blue in each other's channels. Two observers with different + * conventions is what made this take three attempts. */ + packed = (uint32_t)(CLAMP(color->f[0], 0.0f, 1.0f) * 255.0f + 0.5f) << 16 | + (uint32_t)(CLAMP(color->f[1], 0.0f, 1.0f) * 255.0f + 0.5f) << 8 | + (uint32_t)(CLAMP(color->f[2], 0.0f, 1.0f) * 255.0f + 0.5f); + /* A one-byte target keeps the low byte of this, which is blue - so a + * red clear of an eight-bit surface stored nothing and read back as + * zero. The single channel such a surface carries is the red one for + * r8 and the alpha one for a8, and both of them are replicated here + * so whichever byte the back end takes is the right value. The X + * server's glyph atlas is a depth-eight pixmap, and it is cleared + * before anything is drawn into it. */ + if (c->fb.nr_cbufs && c->fb.cbufs[0].texture) { + enum pipe_format cf = c->fb.cbufs[0].format; + + if (cf == PIPE_FORMAT_A8_UNORM || cf == PIPE_FORMAT_R8_UNORM) { + unsigned v = cf == PIPE_FORMAT_A8_UNORM ? + (unsigned)(CLAMP(color->f[3], 0.0f, 1.0f) * + 255.0f + 0.5f) : + (unsigned)(CLAMP(color->f[0], 0.0f, 1.0f) * + 255.0f + 0.5f); + + packed = v | v << 8 | v << 16 | v << 24; + } + } + /* The draws already in the frame were moved into its damage rectangle, + * and a clearing frame renders the whole surface - so they go out + * first rather than being drawn at the origin. */ + if (c->ctx.dmg_on > 0 && c->ctx.draws) + sgx_pipe_submit_pending_why(c, "a clear over a damaged frame"); + sgx_set_clear_enable(&c->ctx, 1); + sgx_set_clear_color(&c->ctx, packed); + sgx_request_clear(&c->ctx); +} + +static void sgx_pipe_set_viewport_states(struct pipe_context *pc, + unsigned start, unsigned n, + const struct pipe_viewport_state *vp) +{ + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + + if (!n || !vp) + return; + /* One frame carries one viewport: ctx_set_mte_viewport() writes it + * into the frame's own state at flush, so whichever is bound last + * would win for every draw already in it. ioquake3's main menu renders + * its logo model into a 640x120 band at the top and its text over the + * whole 640x480, and the logo came out stretched four times down the + * screen - 480 over 120 - when the two shared a frame. */ + /* Off by default since 2026-08-31. This split cost five of the ten + * submits a displayed frame took, each one a tiler pass and a render + * that loads back and stores every tile. It was kept because removing + * it used to take the lightmaps with it - the walls of q3dm1 came out + * flat and untextured - which said the split was holding the frame's + * texture state together rather than its viewport. + * + * Per-record texture state settled that, and the measurement has + * inverted: with the split q3dm1's walls are flat olive blocks and a + * frame holds ten thousand colours, without it they are stone masonry + * and it holds sixty thousand, at 12.8 fps against 16.0. The split is + * now the thing corrupting the frame. SGX_VP_SPLIT puts it back. */ + if (c->ctx.draws && memcmp(&c->viewport, &vp[0], sizeof vp[0]) && + SGX_ENVS("SGX_VP_SPLIT")) { + sgx_dbg("viewport changed with %u draw(s) in the frame; it " + "ends here\n", c->ctx.draws); + sgx_perf_note("viewport"); + sgx_flush(&c->ctx); + sgx_set_clear_enable(&c->ctx, 0); + } + c->viewport = vp[0]; + sgx_set_viewport(&c->ctx, vp[0].scale, vp[0].translate); + draw_set_viewport_states(c->draw, start, n, vp); +} + +/* State the hardware has no path for yet. Accepting it and doing nothing is + * wrong in a way that shows on screen; refusing it is wrong in a way that + * stops the driver being usable at all. These are the second kind: they are + * accepted, and what they would have changed is listed in the driver's + * README rather than left to be discovered. */ +/* Gallium's factors onto the Render operators the hardware's compositing + * shader can carry out. Only the combinations that are one of those operators + * exactly are taken; anything else would have to be approximated, and a + * compositor drawing with the wrong blend is worse off than one told it cannot + * have the blend it asked for. */ +static enum sgx_blend_op sgx_blend_op_of(const struct pipe_blend_state *b) +{ + const struct pipe_rt_blend_state *rt = &b->rt[0]; + + if (!rt->blend_enable) + return SGX_BLEND_NONE; + if (rt->rgb_func != PIPE_BLEND_ADD || rt->alpha_func != PIPE_BLEND_ADD) + return SGX_BLEND_NONE; + + /* Premultiplied over: what every compositor and every Qt scene graph + * asks for. */ + if (rt->rgb_src_factor == PIPE_BLENDFACTOR_ONE && + rt->rgb_dst_factor == PIPE_BLENDFACTOR_INV_SRC_ALPHA) + return SGX_BLEND_OVER; + if (rt->rgb_src_factor == PIPE_BLENDFACTOR_ONE && + rt->rgb_dst_factor == PIPE_BLENDFACTOR_ONE) + return SGX_BLEND_ADD; + if (rt->rgb_src_factor == PIPE_BLENDFACTOR_ONE && + rt->rgb_dst_factor == PIPE_BLENDFACTOR_ZERO) + return SGX_BLEND_SRC; + if (rt->rgb_src_factor == PIPE_BLENDFACTOR_ZERO && + rt->rgb_dst_factor == PIPE_BLENDFACTOR_ONE) + return SGX_BLEND_DST; + if (rt->rgb_src_factor == PIPE_BLENDFACTOR_ZERO && + rt->rgb_dst_factor == PIPE_BLENDFACTOR_ZERO) + return SGX_BLEND_CLEAR; + return SGX_BLEND_NONE; +} + +/* A GL logic op, as far as the blend unit reaches. The unit computes + * src * A + dst * B with A and B selected from {0, src, dst, src.a, dst.a} + * and optionally inverted, so the three operations whose result is a + * constant, the source or the destination fall out of it exactly. The other + * thirteen are bitwise functions of the two operands, which no instruction + * this driver can emit computes - see gl-re/state-model.md section 8, where + * the vendor compiles the logic op into the fragment program. + * + * Returns SGX_BLEND_NONE and clears *ok for those, which leaves the draw + * writing the source: the same thing the driver did for every logic op + * before, and said so nowhere. */ +static enum sgx_blend_op sgx_logicop_op(unsigned func, int *ok) +{ + *ok = 1; + switch (func) { + case PIPE_LOGICOP_CLEAR: return SGX_BLEND_CLEAR; + case PIPE_LOGICOP_COPY: return SGX_BLEND_SRC; + case PIPE_LOGICOP_NOOP: return SGX_BLEND_DST; + default: *ok = 0; return SGX_BLEND_NONE; + } +} + +/* Whether the descriptor builds, with placeholder registers: the words are + * made again with the real ones when the program is assembled. */ +static int sgx_blend_probe(const struct sgx_blend_state *s) +{ + uint64_t insn[SGX_BLEND_MAX_INSNS]; + unsigned n = 0, used = 0; + + /* A placeholder second colour too: whether a dual source state builds + * does not depend on which register holds it, only on there being + * one. The program's own is used when the words are made for real. */ + return sgx_blend_build(&s->desc, 0, 0, 1, 2, SGX_BLEND_MAX_INSNS, + SGX_ENVS("SGX_BLEND_SPLIT") ? SGX_BLEND_F_SPLIT : 0, + insn, SGX_BLEND_MAX_INSNS, &n, &used); +} + +static void *sgx_pipe_create_blend(struct pipe_context *pc, + const struct pipe_blend_state *b) +{ + struct sgx_blend_state *s = CALLOC_STRUCT(sgx_blend_state); + + (void)pc; + if (!s) + return NULL; + if (SGX_ENVS("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: create blend: logicop %u func %u, " + "blend_enable %u, mask 0x%x\n", b->logicop_enable, + b->logicop_func, b->rt[0].blend_enable, + b->rt[0].colormask); + s->op = sgx_blend_op_of(b); + s->colormask = b->rt[0].colormask & 0xfu; + /* A logic op replaces the blend equation rather than joining it, and + * is read from a field of its own that this driver used to ignore + * entirely - every one of the sixteen drew as a plain copy, which is + * right for exactly one of them. */ + if (b->logicop_enable) { + unsigned src_f = 0, dst_f = 0; + int ok = 1; + + s->logicop = 1; + s->logicop_func = b->logicop_func; + /* As blend factors, which is what the unit takes: the result + * is a sum of the source and the destination scaled by one or + * zero, and that expresses exactly the three operations that + * are not bitwise. */ + switch (b->logicop_func) { + case PIPE_LOGICOP_CLEAR: /* 0 */ + src_f = PIPE_BLENDFACTOR_ZERO; + dst_f = PIPE_BLENDFACTOR_ZERO; + break; + case PIPE_LOGICOP_NOOP: /* dst */ + src_f = PIPE_BLENDFACTOR_ZERO; + dst_f = PIPE_BLENDFACTOR_ONE; + break; + case PIPE_LOGICOP_COPY: /* src: no blend at all */ + s->op = SGX_BLEND_SRC; + s->logicop_ok = 1; + return s; + default: + ok = 0; + break; + } + s->logicop_ok = ok; + s->desc.rgb_src = s->desc.alpha_src = (unsigned char)src_f; + s->desc.rgb_dst = s->desc.alpha_dst = (unsigned char)dst_f; + s->desc.rgb_func = s->desc.alpha_func = PIPE_BLEND_ADD; + s->desc.mask = 0xf; + s->desc.enable = 1; + if (ok && !sgx_blend_probe(s)) { + s->have_insn = 1; + s->op = sgx_logicop_op(b->logicop_func, &ok); + } else { + s->op = SGX_BLEND_NONE; + s->have_insn = 0; + fprintf(sgx_log(), "sgx: logic op %u is a bitwise " + "function of source and destination, which " + "this part has no instruction for; the draw " + "writes the source\n", b->logicop_func); + } + return s; + } + /* Built from the factors rather than matched against a named operator; + * the named ones stay as the fallback for a state the builder refuses. + * + * No channel at all is no pixel: the ISP word gets TAGWRITEDIS and + * the program never runs, the vendor's rule for a colour mask of + * zero (validate.c:3624-3631). The stencil-only clear is one. */ + if (!s->colormask) + return s; + /* A partial write mask goes through the blend path too, as a masked + * copy behind the blend (or alone): the channels masked off keep what + * the destination held, and only a read-modify-write draw is handed + * that. ISP word B carries the same mask as its pass key (UPASSCTL), + * which keeps a masked object in a pass of its own; the vendor masks + * in the program too (use.c:784, CreateColorMaskUSECode). */ + s->desc.rgb_src = (unsigned char)b->rt[0].rgb_src_factor; + s->desc.rgb_dst = (unsigned char)b->rt[0].rgb_dst_factor; + s->desc.alpha_src = (unsigned char)b->rt[0].alpha_src_factor; + s->desc.alpha_dst = (unsigned char)b->rt[0].alpha_dst_factor; + s->desc.rgb_func = (unsigned char)b->rt[0].rgb_func; + s->desc.alpha_func = (unsigned char)b->rt[0].alpha_func; + s->desc.mask = (unsigned char)s->colormask; + s->desc.enable = b->rt[0].blend_enable ? 1 : 0; + if (SGX_ENVS("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: create blend: factors %02x/%02x " + "%02x/%02x func %u/%u, src1 %d\n", s->desc.rgb_src, + s->desc.rgb_dst, s->desc.alpha_src, s->desc.alpha_dst, + s->desc.rgb_func, s->desc.alpha_func, + sgx_blend_uses_src1(&s->desc)); + if (s->desc.enable || s->colormask != 0xfu) { + if (!sgx_blend_probe(s)) + s->have_insn = 1; + else if (s->colormask != 0xfu && !s->desc.enable) { + sgx_unsupported("a colour write mask this blend cannot " + "carry is dropped and every channel is " + "written"); + sgx_dbg("colour mask 0x%x cannot be built\n", + s->colormask); + } + } + /* Whether the ISP sorts it as translucent, which a write mask or a + * logic op does not make it. */ + s->translucent = b->rt[0].blend_enable && + sgx_blend_translucent(b->rt[0].rgb_src_factor, + b->rt[0].rgb_dst_factor, + b->rt[0].alpha_src_factor, + b->rt[0].alpha_dst_factor); + if (SGX_ENVS("SGX_DUMP_BLEND")) + fprintf(sgx_log(), "sgx: blend: enable %d rgb %u/%u alpha %u/%u " + "eq %u/%u mask 0x%x -> op %d insn %d\n", + b->rt[0].blend_enable, b->rt[0].rgb_src_factor, + b->rt[0].rgb_dst_factor, b->rt[0].alpha_src_factor, + b->rt[0].alpha_dst_factor, b->rt[0].rgb_func, + b->rt[0].alpha_func, s->colormask, (int)s->op, + s->have_insn); + /* Reported here and not from sgx_blend_op_of(): that is only the + * named-operator match, and the SOP2 path below builds the arbitrary + * pairs it does not name - including SRC_ALPHA/INV_SRC_ALPHA, which is + * most draws in most programs. Saying "not blended" there called every + * one of those a defect. This is the condition that means it: neither + * path produced anything. */ + if (b->rt[0].blend_enable && !s->have_insn && s->op == SGX_BLEND_NONE) { + static int said; + + if (!said) { + said = 1; + fprintf(sgx_log(), "sgx: blend %u/%u alpha %u/%u equation " + "%u/%u has neither a named operator nor an " + "instruction - the draw is not blended and " + "what is drawn will not match what was asked " + "for\n", b->rt[0].rgb_src_factor, + b->rt[0].rgb_dst_factor, + b->rt[0].alpha_src_factor, + b->rt[0].alpha_dst_factor, + b->rt[0].rgb_func, b->rt[0].alpha_func); + } + sgx_dbg("blend: %u/%u with equation %u cannot be built, so the " + "draw is unblended\n", + b->rt[0].rgb_src_factor, b->rt[0].rgb_dst_factor, + b->rt[0].rgb_func); + } + return s; +} + +static void sgx_pipe_bind_blend(struct pipe_context *pc, void *state) +{ + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + + c->cur_blend = state; + sgx_bind_blend(&c->ctx, state); +} + +static void sgx_pipe_set_blend_color(struct pipe_context *pc, + const struct pipe_blend_color *bc) +{ + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + uint32_t packed = sgx_blend_const_pack(bc->color); + + /* The constant is loaded by the blend's own instructions, so a draw + * with a new one cannot share the record before it. */ + if (packed != c->blend_const) + c->blend_const_serial++; + c->blend_const = packed; + sgx_set_blend_color(&c->ctx, packed); +} + +static void sgx_pipe_set_stencil_ref(struct pipe_context *pc, + const struct pipe_stencil_ref ref) +{ + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + + c->stencil_ref = ref; + sgx_pipe_apply_dsa(c); +} + +/* The sample mask. + * + * There is no per-sample write mask on this part. The ISP's multisample + * registers carry positions and nothing else - EUR_CR_ISP_MULTISAMPLECTL is + * four (x, y) pairs and EUR_CR_3D_AA_MODE has one field, ENABLE + * (sgx535defs.h:1238-1241, 1441-1465) - and the vendor's own GLES driver + * confirms it by omission: glSampleCoverage stores its value and its invert + * flag and nothing ever reads them again (opengles2/state.c:908-919 against + * every use of gc->sState.sMultisample, which is the two setters and the two + * getters). + * + * So the state is answered rather than approximated: + * + * every sample selected - the ordinary case, and the only one a mask of ~0 + * produces - is the hardware's own behaviour and costs nothing; + * + * no sample selected must draw nothing. That is exact, and it is honoured + * by dropping the draws while the mask stands. A stub let a mask of zero + * draw everything, which is a wrong answer the caller cannot see; + * + * some samples selected cannot be expressed. It is refused with a + * diagnostic and the draws are dropped rather than drawn: a partial + * coverage mask asks for less than everything, so drawing everything is + * the one answer that is certainly wrong. Nothing this driver exposes + * reaches it - glSampleMaski needs GL 3.2 and this screen reports 2.1, and + * mesa/st only derives a partial mask from glSampleCoverage at a fraction + * strictly between zero and one (st_atom_msaa.c:117-135) - so it is a + * refusal, not a fallback path. + */ +static void sgx_pipe_set_sample_mask(struct pipe_context *pc, unsigned mask) +{ + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + unsigned n = c->fb_samples ? c->fb_samples : 1u; + unsigned full = n >= 32u ? ~0u : (1u << n) - 1u; + unsigned sel = mask & full; + + if (sel && sel != full && c->sample_mask != mask) + fprintf(sgx_log(), "sgx: a sample mask of 0x%x over %u " + "sample(s) selects some and not all, and this part has " + "no per-sample write mask - the draws are dropped " + "rather than drawn whole\n", mask, n); + c->sample_mask = mask; +} + +/* Whether the mask in force writes no fragment. Computed rather than cached, + * so the answer cannot be stale across a framebuffer whose sample count + * differs from the one the mask was set against. */ +static int sgx_sample_mask_blocks(const struct sgx_pipe_context *c) +{ + unsigned n = c->fb_samples ? c->fb_samples : 1u; + unsigned full = n >= 32u ? ~0u : (1u << n) - 1u; + + return (c->sample_mask & full) != full; +} + +/* The scissor rectangle, kept in the screen space the vertex records use. + * + * Gallium gives it in framebuffer coordinates with y upward; this frame's + * viewport transform flips y, so the rows are mirrored here once rather than + * per triangle. Nothing in the reverse-engineered register set is a scissor, + * so it is applied by clipping the geometry - see sgx_vbuf_scissor_tris(). */ +static void sgx_pipe_set_scissor_states(struct pipe_context *pc, + unsigned start, unsigned n, + const struct pipe_scissor_state *sc) +{ + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + + (void)start; + if (n && sc) + c->scissor_state = sc[0]; + if (!n || !sc) { + c->ctx.scissor_on = 0; + return; + } + /* The clipper tests the emitted vertices, which are already in the + * box's own space, so mirroring against the height inverted it - and + * did so against a stale height when the scissor was set first. */ + c->ctx.scissor_x0 = (float)sc[0].minx; + c->ctx.scissor_x1 = (float)sc[0].maxx; + c->ctx.scissor_y0 = (float)sc[0].miny; + c->ctx.scissor_y1 = (float)sc[0].maxy; + c->ctx.scissor_on = 1; +} + +static void sgx_pipe_set_constant_buffer(struct pipe_context *pc, + mesa_shader_stage stage, unsigned index, + const struct pipe_constant_buffer *cb) +{ + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + const void *p = NULL; + unsigned size = 0; + + if (cb) { + if (cb->user_buffer) { + p = cb->user_buffer; + size = cb->buffer_size; + } else if (cb->buffer) { + p = sgx_pipe_resource_cpu(pc, cb->buffer); + p = p ? (const char *)p + cb->buffer_offset : NULL; + size = cb->buffer_size; + } + } + /* The vertex stage's constants go two places: the draw module's copy + * for when it runs the program, and the primary attribute bank for + * when the part does. The fragment stage's go into the secondary + * attribute bank when the frame is built. */ + if (stage == MESA_SHADER_VERTEX && index == 0 && p && size) { + unsigned n = size / 4; + + unsigned want = n; + + if (!sgx_hwtcl_limm_uniforms() && n > SGX_HWTCL_MAX_UNI) + n = SGX_HWTCL_MAX_UNI; + if (n > SGX_HWTCL_MAX_UNI_LIMM) + n = SGX_HWTCL_MAX_UNI_LIMM; + /* Truncating renders with whatever the bank held for the + * uniforms that did not fit, which looks like a shading bug + * rather than a limit being hit. */ + if (want != n) + sgx_dbg("vertex uniforms: %u dwords truncated to %u\n", + want, n); + memcpy(c->vs_const, p, n * 4); + c->vs_const_n = n; + c->vs_const_trunc = want != n; + sgx_set_vs_constants(&c->ctx, c->vs_const, n); + } + /* The fragment stage's uniforms live in the secondary attribute bank, + * which the secondary PDS program loads; nothing carried them before + * and a shader whose colour is a uniform painted with whatever was + * there. */ + if (stage == MESA_SHADER_FRAGMENT && index == 0) { + if (cb) + c->cur_fs_cb = *cb; + else + memset(&c->cur_fs_cb, 0, sizeof c->cur_fs_cb); + sgx_set_fs_constants(&c->ctx, p, p ? size / 4 : 0); + /* Inside the test: the serial opens a new draw record when it + * moves, and bumping it for a vertex-stage upload too started + * one per object. */ + c->fs_const_serial++; + } + /* The draw module runs the vertex stage and the stages that feed it, + * and asserts on anything else - the fragment stage's constants are + * the frame's business, not its. */ + if (stage == MESA_SHADER_VERTEX || stage == MESA_SHADER_GEOMETRY || + stage == MESA_SHADER_TESS_CTRL || stage == MESA_SHADER_TESS_EVAL) { + /* The JIT loads constants as aligned vectors, so an unaligned + * base is not merely slow there. */ + if (getenv("SGX_DUMP_CB")) + fprintf(sgx_log(), "sgx: cb: stage %u slot %u p %p " + "align %u size %u off %u user %d res %u\n", + (unsigned)stage, index, p, + (unsigned)((uintptr_t)p & 15u), size, + cb ? cb->buffer_offset : 0, + cb && cb->user_buffer ? 1 : 0, + (cb && cb->buffer) ? cb->buffer->width0 : 0); + { + /* Experiment: the JIT bounds-checks a constant read against + * this size and returns zero past it, where the interpreter + * does not. A matrix reading as zero is what w = 0 and an + * infinite RHW look like. */ + unsigned sz = size; + const char *e = getenv("SGX_CB_PAD"); + + /* What the stage may read, not what the state tracker says it + * wrote. The JIT bounds-checks every constant read against this + * and returns zero past it, while the interpreter reads on - so + * a shader indexing past buffer_size gets a matrix of zeroes + * under one and the right one under the other, which is every + * vertex through the module arriving as -nan. The buffer is + * ours and the values past the mark are the ones the shader + * wants, so the readable extent is the rest of the resource. */ + if (cb && cb->buffer && !cb->user_buffer && + cb->buffer->width0 > cb->buffer_offset) + sz = cb->buffer->width0 - cb->buffer_offset; + if (e && *e && sz) + sz += (unsigned)atoi(e); + draw_set_mapped_constant_buffer(c->draw, stage, index, p, + sz); + } + } +} + +/* Gallium's swizzle terms in the program's own encoding. Mapped rather than + * cast: the two happen to agree today, and a silent disagreement would show + * up as a wrong colour channel rather than as a build error. */ +static unsigned sgx_swz_chan(unsigned char s) +{ + switch (s) { + case PIPE_SWIZZLE_X: return 0u; + case PIPE_SWIZZLE_Y: return 1u; + case PIPE_SWIZZLE_Z: return 2u; + case PIPE_SWIZZLE_W: return 3u; + case PIPE_SWIZZLE_0: return SGX_SWZ_ZERO; + case PIPE_SWIZZLE_1: return SGX_SWZ_ONE; + default: return SGX_SWZ_ZERO; + } +} + +/* DOUTT0 CHANREPLICATE on AL88, which the vendor sets for A8L8 alongside + * the one-byte formats (sgx535pixfmts.h:603, opengles2/texmgmt.c:3466-3480, + * "to reduce swizzle requirements") and this driver has never set. What the + * unit delivers for U88 with the bit is not in the DDK's text; the lumalpha + * feature case reads it out under both settings. Off until it has. */ +static int sgx_al88_replicate(void) +{ + static int on = -1; + + if (on < 0) { + const char *e = SGX_ENVS("SGX_AL88_REPLICATE"); + + on = e && *e && *e != '0'; + } + return on; +} + +/* Which unpacked slot byte 0 and byte 1 of an AL88 texel arrive in. */ +static void sgx_al88_slots(unsigned *s0, unsigned *s1) +{ + static int a = -1, b = -1; + + if (a < 0) { + const char *e = SGX_ENVS("SGX_AL88_SWZ"); + + if (e && e[0] >= '0' && e[0] <= '3' && e[1] >= '0' && + e[1] <= '3') { + a = e[0] - '0'; + b = e[1] - '0'; + } else if (sgx_al88_replicate()) { + a = 2; + b = 3; + } else { + a = 2; + b = 1; + } + } + *s0 = (unsigned)a; + *s1 = (unsigned)b; +} + +/* Whether the texture unit decodes sRGB before it filters - DOUTT0 GAMMA, + * bit 27 on this core (sgxdefs.h:4344, in the branch that is neither 543/544/ + * 554 nor 545) - or the fragment program decodes the filtered texel as + * before. The DDK says what the bit is for in its own words: + * SGX_FEATURE_GAMMACORRECT_TEXTURES, "Gamma correction is supported on + * texture reads" (sgxfeaturedefs.h:44-46), and SGX535 has the feature + * (:192). A texture read is upstream of the filter, so the bit is the decode + * GL asks for and the program's is the approximation. + * + * SGX_SRGB_SHADER=1 keeps the program's decode, which is the control the + * srgbdec feature case compares against. */ +static int sgx_srgb_hw(void) +{ + static int on = -1; + + if (on < 0) { + const char *e = SGX_ENVS("SGX_SRGB_SHADER"); + + on = !(e && *e && *e != '0'); + } + return on; +} + +/* Whether the unit's decode can serve this view, so that exactly one of the + * two decodes runs. The bit is in DOUTT0, not in the format record, so what + * it applies the curve to is whatever the TAG unpacked; the DDK's own format + * table has no gamma row (sgx535pixfmts.h) and no vendor path sets the bit, + * so the only formats claimed for it here are the ones the unit unpacks as + * four unsigned bytes - the two sRGB rows sgx_format_of() maps. Anything + * else - a block format, a float one - keeps the program's decode rather + * than being handed to a bit whose behaviour on it is not established. + * + * The alpha channel is a separate question the DDK does not answer: GL never + * decodes alpha, the program's decode writes rgb only, and whether the bit + * spares alpha is an experiment. */ +static int sgx_srgb_in_unit(const struct pipe_sampler_view *pv) +{ + if (!pv || !util_format_is_srgb(pv->format) || !sgx_srgb_hw()) + return 0; + return pv->format == PIPE_FORMAT_B8G8R8A8_SRGB || + pv->format == PIPE_FORMAT_R8G8B8A8_SRGB; +} + +/* The border policy, sgx_border_policy() in sgx_resource.c, cached: it is + * asked once per draw by sgx_pipe_settle_borders(). */ +static enum sgx_border_policy sgx_border(void) +{ + static int p = -1; + + if (p < 0) + p = (int)sgx_border_policy(); + return (enum sgx_border_policy)p; +} + +/* Whether a border colour is the one CLAMPBDR was expected to clamp to. The + * part has no border-colour state of any kind - DOUTT0 is allocated bit for + * bit, DOUTT2 is the address, and this core has three texture words where + * later cores have four, whose fourth is a swizzle - so a constant in the + * hardware was all that was left, and transparent black was both GL's default + * and the only value a constant would sensibly be. + * + * Measured 2026-09-03, and the part does not do it: with mode 5 selected and + * no map, the frame comes back with neither the draw nor the clear in it. So + * this comparison no longer decides a mode, it only decides which colour the + * probe asks for; see sgx_wrap_of(). */ +static int sgx_border_is_fixed_colour(const float rgba[4]) +{ + return rgba[0] == 0.0f && rgba[1] == 0.0f && + rgba[2] == 0.0f && rgba[3] == 0.0f; +} + +static void sgx_set_unit_swizzle(struct sgx_pipe_context *c, unsigned unit, + const struct pipe_sampler_view *pv) +{ + unsigned ch[4]; + uint16_t k; + + if (!pv || SGX_ENVS("SGX_NO_TEX_SWIZZLE")) { + k = SGX_SWZ_IDENTITY; + } else { + unsigned i; + + ch[0] = sgx_swz_chan(pv->swizzle_r); + ch[1] = sgx_swz_chan(pv->swizzle_g); + ch[2] = sgx_swz_chan(pv->swizzle_b); + ch[3] = sgx_swz_chan(pv->swizzle_a); + /* The part's one-byte format hands the fetched byte back in + * alpha. A luminance or red-only view names that byte as red, + * which is a channel the fetch never wrote - so every such + * texture sampled black. Whatever channel such a view names, + * there is only the one byte to name. */ + if (sgx_format_of(pv->format) == SGX_FMT_A8) + for (i = 0; i < 4; i++) + if (ch[i] < 4u) + ch[i] = 3u; + /* The two-byte format's bytes land where the same bytes of a + * 32-bit texel would: byte 0 in blue, byte 1 in green - + * measured raw in gl-re/textures.md section 10.1, a U88 + * texel 0x14C8 sampling as (0, 20, 200). A view names them as + * red and green, so both are moved. With CHANREPLICATE set + * (SGX_AL88_REPLICATE) the vendor reads luminance from + * channel 0 and alpha from channel 3, so byte 1 is taken + * from alpha then; SGX_AL88_SWZ= overrides + * either default while the lumalpha case settles it. */ + if (sgx_format_of(pv->format) == SGX_FMT_AL88) { + unsigned s0, s1; + + sgx_al88_slots(&s0, &s1); + for (i = 0; i < 4; i++) + if (ch[i] == 0u) + ch[i] = s0; + else if (ch[i] == 1u) + ch[i] = s1; + } + k = SGX_SWZ_MAKE(ch[0], ch[1], ch[2], ch[3]); + } + + if (SGX_ENVS("SGX_DEBUG") && pv) + sgx_dbg("unit %u view %s (hw %d) swizzle %u%u%u%u -> key %03x\n", + unit, util_format_short_name(pv->format), + (int)sgx_view_format_of(pv->format), pv->swizzle_r, + pv->swizzle_g, pv->swizzle_b, pv->swizzle_a, k); + /* Only grow for a unit that asks for something: an unbound unit and an + * ordinary view are both the identity, which the lookup supplies for + * anything past the end. */ + if (unit >= c->nswz) { + unsigned n, i; + uint16_t *p; + + if (k == SGX_SWZ_IDENTITY) + return; + n = unit + 1u; + p = realloc(c->swz, n * sizeof *p); + if (!p) + return; + for (i = c->nswz; i < n; i++) + p[i] = SGX_SWZ_IDENTITY; + c->swz = p; + c->nswz = n; + } + c->swz[unit] = k; +} + +/* Grow a per-unit key array to hold unit; the entries below it read as + * zero, the identity of every key. Returns 0 when there is no room. */ +static int sgx_unit_key_room(void **arr, unsigned *n, unsigned unit, + size_t size) +{ + void *p; + unsigned want = unit + 1u; + + if (unit < *n) + return 1; + p = realloc(*arr, (size_t)want * size); + if (!p) + return 0; + memset((char *)p + (size_t)*n * size, 0, (size_t)(want - *n) * size); + *arr = p; + *n = want; + return 1; +} + +/* What the unit's fetch delivers, from the view's format: the texel class + * and the plane count, baked into the fragment program. */ +static void sgx_set_unit_cls(struct sgx_pipe_context *c, unsigned unit, + const struct pipe_sampler_view *pv) +{ + unsigned char k = 0; + + if (pv) { + enum sgx_format f = sgx_view_format_of(pv->format); + + k = SGX_TEXKEY(sgx_format_texclass(f), + sgx_format_chunks(pv->format)); + } + if (!k && unit >= c->ncls) + return; + if (!sgx_unit_key_room((void **)&c->cls, &c->ncls, unit, sizeof *c->cls)) + return; + c->cls[unit] = k; +} + +/* The comparison a shadow sampler on this unit performs, with the view + * swizzle it is read through: what nir_lower_tex_shadow is given, so a + * change of either rebuilds the program from its NIR. */ +static void sgx_set_unit_cmp(struct sgx_pipe_context *c, unsigned unit) +{ + const struct sgx_sampler_state *st = unit < SGX_MAX_TEX_UNITS ? + c->cur_samplers[unit] : NULL; + uint32_t k = 0; + + if (st && st->compare) + k = SGX_CMPKEY(st->compare - 1u, + unit < c->nswz ? c->swz[unit] : SGX_SWZ_IDENTITY); + if (!k && unit >= c->ncmp) + return; + if (!sgx_unit_key_room((void **)&c->cmp, &c->ncmp, unit, sizeof *c->cmp)) + return; + c->cmp[unit] = k; +} + +static void sgx_pipe_set_sampler_views(struct pipe_context *pc, + mesa_shader_stage stage, + unsigned start, unsigned n, + unsigned unbind, + struct pipe_sampler_view **views) +{ + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + static int move_va = -1; + unsigned i; + + /* The vertex stage's views go to the draw module, which samples them + * on the CPU. They are not the fragment stage's and share none of the + * unit state below. */ + if (stage == MESA_SHADER_VERTEX) { + for (i = 0; i < n && start + i < SGX_MAX_VTX_SAMPLERS; i++) + c->vtx_views[start + i] = views ? views[i] : NULL; + for (i = 0; i < unbind && start + n + i < SGX_MAX_VTX_SAMPLERS; + i++) + c->vtx_views[start + n + i] = NULL; + c->nvtx_views = 0; + for (i = 0; i < SGX_MAX_VTX_SAMPLERS; i++) + if (c->vtx_views[i]) + c->nvtx_views = i + 1u; + draw_set_sampler_views(c->draw, MESA_SHADER_VERTEX, + c->vtx_views, c->nvtx_views); + return; + } + if (stage != MESA_SHADER_FRAGMENT) + return; + /* The vendor never names a texture address at all: a texture is a + * (buffer, offset) pair and the address is produced by an OFFSET + * relocation at submit time - Xpsb.h:61-88 carries no address field, + * and the same struct describes textures and the destination. Each + * record's PDS data names the texture's own base at ds1[2+2t], so one + * scene holds as many textures as the parameter heap has room for and + * a texture change costs one draw record rather than a whole submit. + * + * So a texture keeps the address it was created at, and each record + * names it. Cycling every texture through one fixed address per unit + * is what made a texture change cost a whole submit - five hundred a + * frame in a game - and what left a record naming whatever landed at + * that address last. SGX_TEX_MOVE_VA restores the old scheme. */ + if (move_va < 0) + move_va = SGX_ENVS("SGX_TEX_MOVE_VA") != NULL; + for (i = 0; i < n; i++) { + struct pipe_sampler_view *pv = views ? views[i] : NULL; + struct sgx_sampler_view v; + + /* The program's decode only where the unit's is not used. */ + if (start + i < 32) { + unsigned bit = 1u << (start + i); + + if (pv && util_format_is_srgb(pv->format) && + !sgx_srgb_in_unit(pv)) + c->srgb_mask |= bit; + else + c->srgb_mask &= ~bit; + if (sgx_depth_in_shader(pv)) + c->depth_mask |= bit; + else + c->depth_mask &= ~bit; + } + sgx_set_unit_swizzle(c, start + i, pv); + sgx_set_unit_cls(c, start + i, pv); + sgx_set_unit_cmp(c, start + i); + if (start + i < SGX_MAX_TEX_UNITS) { + c->cur_views[start + i] = pv; + if (start + i + 1u > c->ncur_views) + c->ncur_views = start + i + 1u; + } + + if (!pv || !pv->texture) { + /* Not just the view: the caller's buffer is still + * bound at the unit's address, and the frame issues + * the unit's fetch whether or not anything is + * described there. Clearing the view alone left the + * descriptor all zeroes with a buffer behind it, and + * the tiler stalled on the fetch - "tiler did not + * finish, events 0", which is what left ioquake3 + * black on its first frame. Releasing puts the + * context's own texture and descriptor back. + * + * Send the frame first, though: the draws already in + * it were made against this texture and the frame + * describes one, so releasing while they are still + * unsent left them sampling the context's own empty + * buffer - ioquake3's menu background came out black + * for it. */ + if (c->ctx.draws) { + unsigned nd = c->ctx.draws; + sgx_perf_note("texture released"); + int fret = sgx_flush(&c->ctx); + + sgx_dbg("view %u: sent %u draw(s) before the " + "texture was let go: %d\n", + start + i, nd, fret); + /* What was just sent has to survive the frame + * that follows, so that one does not clear. + * A flush the caller did not ask for is a + * continuation, not a new picture - without + * this every state change threw away the + * draws before it and only the last of them + * reached the screen. */ + sgx_set_clear_enable(&c->ctx, 0); + } + sgx_release_texture(&c->ctx, start + i); + continue; + } + { + struct sgx_pipe_resource *r = + (struct sgx_pipe_resource *)pv->texture; + + /* The frame relocates the texture address out of its + * own object, so the frame's first texture has to be + * that object. The ones after it keep the address they + * were allocated at and are named by the copy of the + * PDS data their draw record carries, which is what + * lets one frame sample more than one texture. */ + /* The frame relocates its first texture's address out + * of one object, so a texture that arrives once the + * frame has draws in it cannot become that object: + * the draws already there would sample the new one. + * Skipping the rebind instead left the frame sampling + * whatever was bound before - for a copy through a + * scratch pixmap, the destination itself, which is + * why a moved window came out in the root colour. + * Send what the frame holds first, exactly as a + * change of render target does. */ + /* A resource is already bound at an address of its own, + * and a draw record names the texture address in its + * own copy of the PDS data - so a second texture does + * not have to displace the first, and the frame does + * not have to end. Moving every texture onto the + * unit's fixed address is what made a texture change + * cost a whole frame: ioquake3 spent one full-screen + * submit per draw, five hundred of them a frame. */ + if (!move_va && r->r.bo.gpu_va && + start + i < SGX_MAX_TEX_UNITS && + c->ctx.tex_bo[start + i] != &r->r.bo) { + /* The submit names every texture the frame + * samples and the kernel takes a bounded + * number, so the frame ends when that budget + * is spent - not when a texture changes. */ + if (c->ctx.draws && + sgx_frame_add_tex(&c->ctx, &r->r.bo)) { + unsigned nd = c->ctx.draws; + sgx_perf_note("texture budget"); + int fret = sgx_flush(&c->ctx); + + sgx_dbg("view %u: sent %u draw(s), the " + "frame was full of textures: " + "%d\n", start + i, nd, fret); + sgx_set_clear_enable(&c->ctx, 0); + sgx_frame_add_tex(&c->ctx, &r->r.bo); + } + sgx_dbg("view %u: refresh %d\n", start + i, + sgx_refresh_texture(&c->ctx, + &r->r.bo)); + sgx_note_texture(&c->ctx, start + i, &r->r.bo); + sgx_dbg("view %u: kept at 0x%llx\n", start + i, + (unsigned long long)r->r.bo.gpu_va); + } else if (c->ctx.draws && start + i < 2 && + c->ctx.tex_bo[start + i] != &r->r.bo) { + unsigned nd = c->ctx.draws; + sgx_perf_note("texture change"); + int fret = sgx_flush(&c->ctx); + + sgx_dbg("view %u: flushed %u draw(s) before a " + "new texture: %d\n", start + i, nd, + fret); + /* What was just sent has to survive the frame + * that follows, so that one does not clear. + * A flush the caller did not ask for is a + * continuation, not a new picture - without + * this every state change threw away the + * draws before it and only the last of them + * reached the screen. */ + sgx_set_clear_enable(&c->ctx, 0); + } + /* Only when the frame is empty does the unit's fixed + * address still have to hold it: the frame's own + * relocation names that object for the first record. */ + if ((move_va || !r->r.bo.gpu_va) && + sgx_use_texture(&c->ctx, start + i, &r->r.bo)) { + sgx_dbg("view %u: the texture would not take " + "the unit's address\n", start + i); + continue; + } + c->cur_tex = pv->texture; + memset(&v, 0, sizeof(v)); + /* The texels, behind the border map and its guard if + * there is one - level 0's own offset either way. */ + v.gpu_va = (uint32_t)r->r.bo.gpu_va + + r->r.border_guard + r->r.border_map; + /* The depth surface is sampled as the packed dword it + * holds; its stride is already the four-byte rule's + * (sgx_resource_create). */ + v.format = (r->r.bind & SGX_BIND_DEPTH_STENCIL) ? + sgx_view_format_of(pv->format) : r->r.format; + v.width = r->r.width; + v.height = r->r.height; + v.stride = r->r.stride; + v.twiddled = r->r.twiddled; + v.depth = r->r.depth; + v.nchunks = r->r.nchunks; + v.chunk_size = r->r.chunk_size; + v.border_map = r->r.border_map; + v.border_base = sgx_border() == SGX_BORDER_MAP_TEXELS || + sgx_border() == SGX_BORDER_BDR_TEXELS; + v.srgb = sgx_srgb_in_unit(pv); + v.chanrep = r->r.format == SGX_FMT_AL88 && + sgx_al88_replicate(); + /* A twiddled chain is described to the hardware. A + * linear one is laid out the vendor's way now - the + * STRIDE rule in sgx_resource_level_texels() - which + * accounts for two of the four level divergences that + * were measured against the blob's layout and not for + * the other two, so it is opt-in under SGX_LINEAR_MIPS + * until the part has been asked; a non-power-of-two + * texture otherwise samples its base level. The levels + * are still allocated and uploaded, so nothing else + * has to know. + * + * The hardware has no first-level field, so a view + * that starts above level 0 is described by pointing + * at that level and shortening the chain. */ + v.levels_undescribed = r->r.nlevels > 1 && + !r->r.twiddled && + !sgx_linear_mips(); + if (r->r.nlevels > 1 && + (r->r.twiddled || sgx_linear_mips()) && + pv->u.tex.first_level < r->r.nlevels) { + unsigned f = pv->u.tex.first_level; + unsigned l = pv->u.tex.last_level < r->r.nlevels + ? pv->u.tex.last_level + : r->r.nlevels - 1u; + + if (l < f) + l = f; + v.gpu_va += r->r.level[f].offset - + r->r.level[0].offset; + v.width = r->r.level[f].width; + v.height = r->r.level[f].height; + v.stride = r->r.level[f].stride; + v.depth = r->r.level[f].depth > 1 ? + r->r.level[f].depth : 0u; + v.nlevels = l - f + 1u; + /* The map's rings for the smaller levels sit + * at its start, not at a fixed distance ahead + * of level f, so a view from a lower level + * has no map the unit could be pointed at. */ + if (f) + v.border_map = 0; + } + { + int ret; + /* A view has to describe no more than the + * object behind it. Nothing checked, and a + * sample of the rows past the end faults the + * MMU with the texture cache as requestor - + * which is what ioquake3 does once it starts + * minifying. Clamp to what is there and say + * so; the descriptor that asked for more is + * the defect to find. */ + uint64_t off = v.gpu_va - r->r.bo.gpu_va; + uint64_t have = r->r.bo.size > off ? + r->r.bo.size - off : 0; + uint64_t want = (uint64_t)v.height * v.stride; + + /* Only when the object's size is known and the + * clamp leaves something to sample: a shared + * or imported object carries no size here, and + * clamping that to nothing renders nothing. */ + if (v.stride && r->r.bo.size && want > have && + have >= v.stride) { + unsigned rows = + (unsigned)(have / v.stride); + + sgx_dbg("view %u: %ux%u at stride %u " + "needs %llu byte(s) and the " + "object holds %llu - clamped " + "to %u row(s)\n", start + i, + v.width, v.height, v.stride, + (unsigned long long)want, + (unsigned long long)have, + rows); + v.height = rows; + } + ret = sgx_set_sampler_view(&c->ctx, + start + i, &v); + if (start + i < SGX_MAX_TEX_UNITS) + c->view_base[start + i] = + r->r.bo.gpu_va; + + sgx_dbg("view %u: va 0x%08x fmt %d %ux%u " + "stride %u levels %u tw %u size %llu " + "-> %d\n", + start + i, v.gpu_va, (int)v.format, + v.width, v.height, v.stride, + v.nlevels, v.twiddled, + (unsigned long long)r->r.bo.size, ret); + } + } + } + for (i = 0; i < unbind; i++) + sgx_set_sampler_view(&c->ctx, start + n + i, NULL); +} + +static struct pipe_sampler_view * +sgx_pipe_create_sampler_view(struct pipe_context *pc, + struct pipe_resource *texture, + const struct pipe_sampler_view *templ) +{ + struct pipe_sampler_view *v = CALLOC_STRUCT(pipe_sampler_view); + + (void)pc; + if (!v) + return NULL; + *v = *templ; + v->reference.count = 1; + v->texture = NULL; + pipe_resource_reference(&v->texture, texture); + v->context = pc; + return v; +} + +static void sgx_pipe_sampler_view_destroy(struct pipe_context *pc, + struct pipe_sampler_view *v) +{ + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + unsigned i; + + /* Same reason as the state objects above: the blitter would reference + * a view that no longer exists. */ + if (c) + for (i = 0; i < SGX_MAX_TEX_UNITS; i++) + if (c->cur_views[i] == v) + c->cur_views[i] = NULL; + pipe_resource_reference(&v->texture, NULL); + FREE(v); +} + +/* Whether to program the texture unit's anisotropic ratio. The field and the + * filter modes are the DDK's for this core and carry no feature gate, but no + * vendor code path writes a ratio - the transfer queue selects the ANISO + * filter mode and then forces the ratio to none - so this has never been seen + * to work on silicon. SGX_ANISO=1 takes it up. */ +static int sgx_aniso_enabled(void) +{ + static int on = -1; + + if (on < 0) { + const char *e = SGX_ENVS("SGX_ANISO"); + + on = e && *e && *e != '0'; + } + return on; +} + +/* Whether a linear (non-power-of-two) chain is described to the unit at all. + * The layout is the vendor's STRIDE rule now, but two of the four level + * divergences measured against the old layout are not explained by it, so the + * chain stays unannounced - level 0 only, as before - until the part has been + * asked with SGX_LINEAR_MIPS=1 (work/feat-mipmap/TESTPLAN.md). */ +static int sgx_linear_mips(void) +{ + static int on = -1; + + if (on < 0) { + const char *e = SGX_ENVS("SGX_LINEAR_MIPS"); + + on = e && *e && *e != '0'; + } + return on; +} + +/* GL_MIRROR_CLAMP_TO_EDGE is a real mode here rather than an approximation - + * it is the unit's FLIPCLAMP - so it no longer collapses onto clamp-to-edge. + * + * GL_CLAMP does not get the mode that bears its name. The address-mode field + * has an OGLCLAMP encoding, code 7, and mapping GL_CLAMP to it renders + * nothing at all: the draw is submitted and accepted and the target comes + * back without even the clear. No vendor code selects it either - the DDK + * defines the constant and never uses it - so code 7 is treated here as an + * encoding this part does not implement, and GL_CLAMP takes the ordinary + * clamp instead. That is what GL_CLAMP degrades to without a border colour, + * and it is what every driver lacking border support does. + * + * GL_CLAMP_TO_BORDER has two forms on this part and neither is free. CLAMPBDR + * (5) reads no map, so it costs nothing and works on any shape, but the colour + * it clamps to is fixed in the hardware; CLAMPBDRMEM (6) reads a border map + * the resource layer lays out in front of a square twiddled texture, which is + * any colour but only that shape and only under SGX_BORDER - a texture without + * a map is caught at draw time by sgx_pipe_settle_borders(). + * GL_MIRROR_CLAMP_TO_BORDER has no form at all and stays an approximation. */ +static enum sgx_wrap sgx_wrap_of(unsigned pipe_wrap, int fixed_colour) +{ + switch (pipe_wrap) { + case PIPE_TEX_WRAP_REPEAT: return SGX_WRAP_REPEAT; + case PIPE_TEX_WRAP_MIRROR_REPEAT: return SGX_WRAP_MIRROR_REPEAT; + case PIPE_TEX_WRAP_MIRROR_CLAMP: + case PIPE_TEX_WRAP_MIRROR_CLAMP_TO_EDGE: + /* No mode mirrors and reads a border: the border codes 4-6 have no + * FLIP form (sgxdefs.h:4407-4414). Mirror-clamp-to-border keeps the + * mirror and loses the border, which is what the diagnostic below + * says; falling to the default lost the mirror as well. */ + case PIPE_TEX_WRAP_MIRROR_CLAMP_TO_BORDER: + return SGX_WRAP_MIRROR_CLAMP; + case PIPE_TEX_WRAP_CLAMP_TO_BORDER: + switch (sgx_border()) { + /* The map-free mode, which renders nothing on this part: the + * probe that measured that, still asking for the colour it + * was expected to supply. */ + case SGX_BORDER_FIXED: + return fixed_colour ? SGX_WRAP_CLAMP_TO_BORDER_FIXED : + SGX_WRAP_CLAMP_TO_EDGE; + /* The same mode with a map in front of the texture. Measured + * 2026-09-03: indistinguishable from CLAMPBDRMEM at both + * ends, which is how we know neither code is missing. */ + case SGX_BORDER_BDR_BASE: + case SGX_BORDER_BDR_TEXELS: + return SGX_WRAP_CLAMP_TO_BORDER_FIXED_MAP; + case SGX_BORDER_MAP_BASE: + case SGX_BORDER_MAP_TEXELS: + return SGX_WRAP_CLAMP_TO_BORDER; + default: + return SGX_WRAP_CLAMP_TO_EDGE; + } + default: return SGX_WRAP_CLAMP_TO_EDGE; + } +} + +static void *sgx_pipe_create_sampler_state(struct pipe_context *pc, + const struct pipe_sampler_state *st) +{ + struct sgx_sampler_state *s = CALLOC_STRUCT(sgx_sampler_state); + int fixed; + + (void)pc; + if (!s) + return NULL; + s->min_filter = st->min_img_filter == PIPE_TEX_FILTER_LINEAR ? + SGX_FILTER_LINEAR : SGX_FILTER_NEAREST; + s->mag_filter = st->mag_img_filter == PIPE_TEX_FILTER_LINEAR ? + SGX_FILTER_LINEAR : SGX_FILTER_NEAREST; + s->mip_filter = st->min_mip_filter == PIPE_TEX_MIPFILTER_LINEAR ? + SGX_MIPFILTER_LINEAR : + st->min_mip_filter == PIPE_TEX_MIPFILTER_NEAREST ? + SGX_MIPFILTER_NEAREST : SGX_MIPFILTER_NONE; + /* Gallium's numbering is not the hardware's, and casting one to the + * other made CLAMP_TO_EDGE into MIRROR_REPEAT. The part has three + * modes; the ones it does not have are refused by taking the nearest + * it does, which is what every driver without a border colour does. */ + fixed = sgx_border_is_fixed_colour(st->border_color.f); + s->wrap_s = sgx_wrap_of(st->wrap_s, fixed); + s->wrap_t = sgx_wrap_of(st->wrap_t, fixed); + s->wrap_r = sgx_wrap_of(st->wrap_r, fixed); + memcpy(s->border_color, st->border_color.f, sizeof s->border_color); + if ((sgx_border() == SGX_BORDER_EDGE || + (sgx_border() == SGX_BORDER_FIXED && !fixed)) && + (st->wrap_s == PIPE_TEX_WRAP_CLAMP_TO_BORDER || + st->wrap_t == PIPE_TEX_WRAP_CLAMP_TO_BORDER || + st->wrap_r == PIPE_TEX_WRAP_CLAMP_TO_BORDER)) + sgx_unsupported(sgx_border() == SGX_BORDER_EDGE ? + "a border-colour wrap mode is sampled as clamp " + "to edge (SGX_BORDER=1 lays out the map)" : + "a border colour the map-free mode cannot " + "supply is sampled as clamp to edge"); + /* The one mode measured to produce no frame at all. It stays + * reachable because it is the probe, but never silently. */ + if (sgx_border() == SGX_BORDER_FIXED && fixed && + (st->wrap_s == PIPE_TEX_WRAP_CLAMP_TO_BORDER || + st->wrap_t == PIPE_TEX_WRAP_CLAMP_TO_BORDER || + st->wrap_r == PIPE_TEX_WRAP_CLAMP_TO_BORDER)) + sgx_unsupported("SGX_BORDER=fixed selects CLAMPBDR without a " + "map, which was measured returning no frame at " + "all on this part - it is a probe, not a mode"); + if (st->wrap_s == PIPE_TEX_WRAP_MIRROR_CLAMP_TO_BORDER || + st->wrap_t == PIPE_TEX_WRAP_MIRROR_CLAMP_TO_BORDER || + st->wrap_r == PIPE_TEX_WRAP_MIRROR_CLAMP_TO_BORDER) + sgx_unsupported("mirror-clamp-to-border has no address mode " + "on this part and is sampled as mirror clamp " + "to edge"); + /* The comparison is not in the unit on this core: it is lowered into + * the fragment program from this, keyed per unit. */ + s->compare = st->compare_mode != PIPE_TEX_COMPARE_NONE ? + (unsigned)st->compare_func + 1u : 0u; + /* The unit has a LOD adjust and a maximum-level clamp, so both are + * carried rather than dropped. There is no minimum-LOD field on this + * part, so st->min_lod cannot be honoured; GL asks for a clamp the + * hardware does not have and the level it would have skipped is + * sampled instead. */ + s->lod_bias = st->lod_bias; + s->max_lod = st->max_lod; + /* Off unless the caller asks and SGX_ANISO allows it. The ratio field + * and the anisotropic filter modes are in the hardware and the DDK + * names them for this core, but no vendor code path ever writes a + * ratio - the one place that selects the filter mode forces the ratio + * to none - so nothing here has been seen working on silicon. Opt-in + * until it is. */ + if (sgx_aniso_enabled()) + s->max_anisotropy = st->max_anisotropy; + return s; +} + +static void sgx_pipe_bind_sampler_states(struct pipe_context *pc, + mesa_shader_stage stage, + unsigned start, unsigned n, + void **states) +{ + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + unsigned i; + + if (stage == MESA_SHADER_VERTEX) { + for (i = 0; i < n && start + i < SGX_MAX_VTX_SAMPLERS; i++) + c->vtx_samplers[start + i] = states ? states[i] : NULL; + c->nvtx_samplers = 0; + for (i = 0; i < SGX_MAX_VTX_SAMPLERS; i++) + if (c->vtx_samplers[i]) + c->nvtx_samplers = i + 1u; + draw_set_samplers(c->draw, MESA_SHADER_VERTEX, + (struct pipe_sampler_state **) + c->vtx_samplers, c->nvtx_samplers); + return; + } + if (stage != MESA_SHADER_FRAGMENT) + return; + for (i = 0; i < n; i++) { + if (start + i < SGX_MAX_TEX_UNITS) { + c->cur_samplers[start + i] = states ? states[i] : NULL; + c->border_sub[start + i] = 0; + if (start + i + 1u > c->ncur_samplers) + c->ncur_samplers = start + i + 1u; + } + sgx_set_unit_cmp(c, start + i); + sgx_bind_sampler_state(&c->ctx, start + i, + states ? states[i] : NULL); + } +} + +/* A border-mode sampler needs the map its texture was laid out with, and + * the map needs the sampler's colour - two objects that only meet at a draw. + * A texture without a map (not square, not a power of two, not twiddled, or + * viewed from a lower level) is sampled clamp-to-edge and said so, exactly + * as the whole mode was before; one with a map gets it filled here. */ +static void sgx_pipe_settle_borders(struct sgx_pipe_context *c) +{ + unsigned u; + + /* Only the map modes have anything to settle: CLAMPBDR reads no map, + * so sgx_sampler_uses_border() is false for it and no fill is due. */ + if (sgx_border() < SGX_BORDER_MAP_BASE) + return; + for (u = 0; u < c->ncur_samplers && u < SGX_MAX_TEX_UNITS; u++) { + const struct sgx_sampler_state *st = + (const struct sgx_sampler_state *)c->cur_samplers[u]; + struct pipe_sampler_view *pv = u < c->ncur_views ? + c->cur_views[u] : NULL; + struct sgx_pipe_resource *r = pv ? + (struct sgx_pipe_resource *)pv->texture : NULL; + int has_map; + + if (!st || !sgx_sampler_uses_border(st)) + continue; + has_map = r && r->r.border_map && !pv->u.tex.first_level; + if (has_map) { + enum sgx_resource_status rs = + sgx_resource_fill_border(&r->r, c->ctx.ws, + st->border_color); + + if (rs != SGX_RESOURCE_OK) + sgx_dbg("unit %u: border map not filled: %s\n", + u, sgx_resource_status_name(rs)); + if (c->border_sub[u]) { + c->border_sub[u] = 0; + sgx_bind_sampler_state(&c->ctx, u, st); + } + } else if (!c->border_sub[u]) { + struct sgx_sampler_state edge = *st; + + if (sgx_wrap_needs_map(edge.wrap_s)) + edge.wrap_s = SGX_WRAP_CLAMP_TO_EDGE; + if (sgx_wrap_needs_map(edge.wrap_t)) + edge.wrap_t = SGX_WRAP_CLAMP_TO_EDGE; + if (sgx_wrap_needs_map(edge.wrap_r)) + edge.wrap_r = SGX_WRAP_CLAMP_TO_EDGE; + sgx_unsupported("a border-colour wrap on a texture " + "without a border map - not square, " + "power of two and twiddled - is sampled " + "as clamp to edge"); + c->border_sub[u] = 1; + sgx_bind_sampler_state(&c->ctx, u, &edge); + } + } +} + +/* --- transfers ---------------------------------------------------------- */ + +/* Whether the fourth byte of a texel of this resource is the alpha the + * missing-alpha fill may write. True for the packed 8888 codes and nothing + * else: an F32 or F1616 texel is four bytes as well, and writing 0xff over + * its top byte turned 0.5 into -1.7e38 - which is what rendered the + * single-plane float textures black. */ +static int sgx_format_has_alpha_byte(const struct sgx_resource *r) +{ + return (r->format == SGX_FMT_A8R8G8B8 || + r->format == SGX_FMT_A8B8G8R8) && + (!r->nchunks || r->nchunks == 1); +} + + +static void sgx_pipe_submit_pending(struct sgx_pipe_context *c); +static void sgx_pipe_submit_pending_why(struct sgx_pipe_context *c, + const char *why); + +/* A texture transfer. The level's own mapping is kept so the unmap can walk + * exactly the level - the staging copy of a twiddled one, the padded rows of + * a linear one - rather than the whole object. */ + +static void *sgx_pipe_texture_map(struct pipe_context *pc, + struct pipe_resource *pres, unsigned level, + unsigned usage, const struct pipe_box *box, + struct pipe_transfer **out) +{ + sgx_xfer_texmap++; + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + struct sgx_pipe_resource *r = (struct sgx_pipe_resource *)pres; + struct sgx_transfer *t; + uint32_t lstride = 0; + unsigned face = r && r->r.nfaces ? (unsigned)box->z : 0u; + /* box->z is a cube's face or a volume's slice; a volume's level is + * staged whole, so its slices are all there. */ + unsigned slice = r && r->r.depth ? (unsigned)box->z : 0u; + void *p = NULL; + + /* One face of a cube at a time: its faces are separate chains, and a + * twiddled face has one staging copy. mesa/st maps a cube a face at a + * time (st_texture_image_map, depth 1), so refusing more is refusing + * nothing it does. */ + if (r && r->r.nfaces && (box->depth > 1 || face >= r->r.nfaces)) { + sgx_dbg("texmap: cube face %u depth %d refused\n", face, + box->depth); + return NULL; + } + if (!r || sgx_resource_map_face_level(&r->r, c->ctx.ws, face, level, + &p, &lstride) != SGX_RESOURCE_OK) + return NULL; + /* Reading a surface must see the draws that were meant to be in it. + * Without this a glReadPixels returned whatever the buffer held from + * the frame before - which is why a scene whose model rendered + * correctly still validated as black. After the map, not before: a + * map that cannot be served was making every frame pay a synchronous + * submit for a read that never happened. */ + /* On a write as much as on a read. The frame holds draws that have + * been accumulated and not yet submitted, and they are meant to have + * happened before whatever the caller is about to do to the surface - + * so leaving them pending puts them *after* it. That is how twm's + * menu lost its text: the X server filled the menu background through + * the part, drew the text into the same pixmap with the CPU, and the + * fill it had asked for first landed last and covered it. + * + * Then wait. The map above waits for whatever was already + * outstanding, which is not the frame submitted on the line before + * it: a deferred submit returns as soon as the render starts. Both + * halves were free while the submit was synchronous. */ + if (usage & (PIPE_MAP_READ | PIPE_MAP_WRITE)) { + sgx_pipe_submit_pending_why(c, "upload: sgx_pipe_texture_map"); + sgx_wait_idle(c->ctx.ws); + } + sgx_dbg("texmap: %ux%u bind 0x%x face %u level %u of %u, stride %u, " + "tw %u, box %dx%d at %d,%d, usage 0x%x\n", + r->r.width, r->r.height, r->r.bind, face, level, + r->r.nlevels, lstride, r->r.twiddled, + box->width, box->height, box->x, box->y, usage); + t = CALLOC_STRUCT(sgx_transfer); + if (!t) { + sgx_resource_unmap_level(&r->r); + return NULL; + } + t->b.b.resource = NULL; + pipe_resource_reference(&t->b.b.resource, pres); + t->b.b.level = level; + t->b.b.usage = usage; + t->b.b.box = *box; + if (!lstride) + lstride = r->r.stride ? r->r.stride : pres->width0; + t->b.b.stride = lstride; + /* A cube's faces are a face stride apart in the object, but the + * pointer handed back is one face's level, so the caller cannot step + * to the next; the depth check above makes sure it does not try. */ + t->b.b.layer_stride = r->r.nfaces ? r->r.face_stride : + t->b.b.stride * + (r->r.depth && level < r->r.nlevels ? + r->r.level[level].height : pres->height0); + t->face = face; + t->level_base = p; + t->level_stride = lstride; + t->level_rows = r->r.nlevels ? r->r.level[level].height : + r->r.height; + if (r->r.depth && level < r->r.nlevels) + t->level_rows *= r->r.level[level].depth; + *out = &t->b.b; + /* A depth resource is one float a pixel however the caller named + * it, so the CPU sees it through a copy in its own format: what a + * glTexImage2D of GL_DEPTH_COMPONENT writes and glReadPixels reads. + * Depth only; the stencil byte of a packed format reads as zero. */ + if (r->zfmt != PIPE_FORMAT_NONE) { + unsigned bs = util_format_get_blocksize(r->zfmt); + size_t need = (size_t)bs * box->width * box->height; + + if (r->zstage_size < need) { + void *z = realloc(r->zstage, need); + + if (!z) { + sgx_resource_unmap_level(&r->r); + pipe_resource_reference(&t->b.b.resource, NULL); + FREE(t); + return NULL; + } + r->zstage = z; + r->zstage_size = (unsigned)need; + } + if ((unsigned)box->width > r->zrow_n) { + float *f = realloc(r->zrow, + (size_t)box->width * sizeof *f); + + if (!f) { + sgx_resource_unmap_level(&r->r); + pipe_resource_reference(&t->b.b.resource, NULL); + FREE(t); + return NULL; + } + r->zrow = f; + r->zrow_n = (unsigned)box->width; + } + if (usage & PIPE_MAP_READ) { + int y; + + /* Out of the form the ISP stored, then into the one + * the caller asked for: reading the surface as floats + * is what made every depth read back as nothing. */ + for (y = 0; y < box->height; y++) { + const char *src = (const char *)p + + (size_t)(box->y + y) * lstride + + (size_t)box->x * 4u; + + util_format_unpack_z_float(SGX_DEPTH_SURF_FMT, + r->zrow, src, box->width); + util_format_pack_z_float(r->zfmt, + (char *)r->zstage + + (size_t)y * bs * box->width, + r->zrow, box->width); + } + } else if (!(usage & PIPE_MAP_WRITE)) { + memset(r->zstage, 0, need); + } + r->zbox = *box; + r->zlevel = level; + t->b.b.stride = bs * box->width; + t->b.b.layer_stride = t->b.b.stride * box->height; + t->level_base = NULL; + return r->zstage; + } + { + unsigned bw, bh, bb, bpp = r->r.format != SGX_FMT_NONE ? + xpsb_format_bpp((uint32_t)r->r.format) : 1; + + /* A block format's box is in texels; its rows are blocks. */ + if (sgx_format_block(r->r.format, &bw, &bh, &bb)) + return (char *)p + + (size_t)(box->y / bh) * t->b.b.stride + + (size_t)(box->x / bw) * bb; + /* The staging copy interleaves the planes. */ + if (r->r.nchunks > 1) + bpp = sgx_resource_texel_bytes(&r->r); + return (char *)p + (size_t)slice * t->b.b.layer_stride + + (size_t)box->y * t->b.b.stride + + (size_t)box->x * (bpp ? bpp : 1); + } +} + +static void sgx_pipe_texture_unmap(struct pipe_context *pc, + struct pipe_transfer *pt) +{ + struct sgx_transfer *t = (struct sgx_transfer *)pt; + struct sgx_pipe_resource *r = (struct sgx_pipe_resource *)pt->resource; + + (void)pc; + /* Whether anything was actually written into a big surface: a display + * server that renders on the CPU leaves its result here, so a surface + * that unmaps still empty says the drawing never happened. */ + if (r && sgx_debug() && r->r.map && r->r.size >= (1u << 20)) { + const uint32_t *px = r->r.map; + unsigned i, n = r->r.size / 4, nz = 0; + + for (i = 0; i < n && i < 65536; i++) + if (px[i] & 0x00ffffffu) + nz++; + sgx_dbg("texunmap: %ux%u, %u of the first 65536 dwords are " + "non-black, first 0x%08x\n", r->r.width, r->r.height, + nz, px[0]); + } + /* The caller's format has no alpha, so whatever it left in that byte + * is not alpha - and GL says the texture samples one there. Written + * before the surface goes back, because the state tracker maps and + * writes these directly rather than going through texture_subdata. + * Without it an RGB texture blended by its source alpha disappears: + * ioquake3's cinematics played as a black rectangle. */ + if (SGX_ENVS("SGX_DEBUG") && r && r->no_alpha) + fprintf(sgx_log(), "sgx: unmap: no-alpha resource, map %p bpp %u " + "usage 0x%x box %dx%d\n", r->r.map, + xpsb_format_bpp((uint32_t)r->r.format), pt->usage, + pt->box.width, pt->box.height); + if (r && r->no_alpha && t->level_base && sgx_format_has_alpha_byte(&r->r) && + (pt->usage & PIPE_MAP_WRITE)) { + /* Every fourth byte of the level that was mapped, before it + * goes back: for a twiddled level that is the staging copy, + * which the unmap below swizzles into the object - filling + * the object first, as this used to, was undone by that + * swizzle and left the level with whatever alpha the state + * tracker wrote for a format that has none. */ + unsigned char *px = t->level_base; + size_t i, n = (size_t)t->level_stride * t->level_rows; + + for (i = 3; i < n; i += 4) + px[i] = 0xff; + if (SGX_ENVS("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: unmap: filled alpha over %u " + "bytes of face %u level %u\n", (unsigned)n, + t->face, pt->level); + } + /* And back the same way round, into the form the ISP stores. */ + if (r && r->zfmt != PIPE_FORMAT_NONE && r->zstage && r->r.map && + r->zrow && (unsigned)r->zbox.width <= r->zrow_n && + (pt->usage & PIPE_MAP_WRITE)) { + unsigned bs = util_format_get_blocksize(r->zfmt); + uint32_t stride = r->r.stride ? r->r.stride : r->r.width * 4u; + int y; + + for (y = 0; y < r->zbox.height; y++) { + char *dst = (char *)r->r.map + + (size_t)(r->zbox.y + y) * stride + + (size_t)r->zbox.x * 4u; + + util_format_unpack_z_float(r->zfmt, r->zrow, + (const char *)r->zstage + + (size_t)y * bs * r->zbox.width, r->zbox.width); + util_format_pack_z_float(SGX_DEPTH_SURF_FMT, dst, + r->zrow, r->zbox.width); + } + } + if (r) + sgx_resource_unmap_level(&r->r); + pipe_resource_reference(&pt->resource, NULL); + FREE(t); +} + +static void sgx_pipe_buffer_subdata(struct pipe_context *pc, + struct pipe_resource *pres, unsigned usage, + unsigned offset, unsigned size, + const void *data) +{ + sgx_xfer_bufsub++; + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + struct sgx_pipe_resource *r = (struct sgx_pipe_resource *)pres; + void *p = NULL; + + (void)usage; + if (!r || sgx_resource_map(&r->r, c->ctx.ws, &p) != SGX_RESOURCE_OK) + return; + if ((uint64_t)offset + size <= r->r.size) + memcpy((char *)p + offset, data, size); + sgx_pipe_shadow_dirty(pres); + sgx_resource_unmap(&r->r); +} + +static void sgx_pipe_texture_subdata(struct pipe_context *pc, + struct pipe_resource *pres, unsigned level, + unsigned usage, const struct pipe_box *box, + const void *data, unsigned stride, + uintptr_t layer_stride) +{ + sgx_xfer_texsub++; + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + struct sgx_pipe_resource *r = (struct sgx_pipe_resource *)pres; + /* The CPU's texel: every plane's, or a block's. */ + unsigned bpp = r ? sgx_resource_texel_bytes(&r->r) : 0; + unsigned nfaces = r ? (r->r.nfaces ? r->r.nfaces : 1u) : 0u; + unsigned bw = 1, bh = 1, z, faces; + int y; + + (void)usage; + if (!r || !bpp || r->zfmt != PIPE_FORMAT_NONE) + return; + sgx_format_block(r->r.format, &bw, &bh, NULL); + /* box->z and depth walk a cube's faces; anything else is one layer + * deep. A face is mapped on its own because a twiddled one has a + * single staging copy. */ + faces = r->r.nfaces ? (unsigned)box->depth : 1u; + if ((unsigned)box->z + faces > nfaces) + return; + for (z = 0; z < faces; z++) { + uint32_t dstride = 0; + void *p = NULL; + const char *src = (const char *)data + (size_t)z * layer_stride; + + if (sgx_resource_map_face_level(&r->r, c->ctx.ws, + (uint32_t)box->z + z, level, + &p, &dstride) != + SGX_RESOURCE_OK) + return; + if (!dstride) + dstride = r->r.stride; + /* A volume's slices are the level's height apart in the + * staging copy; box->z and depth walk them. */ + size_t dslice = r->r.depth && level < r->r.nlevels ? + (size_t)dstride * r->r.level[level].height : 0; + int slices = r->r.depth ? box->depth : 1; + + sgx_dbg("subdata: face %u level %u box %d,%d,%d %dx%dx%d " + "stride %u -> off %u\n", (unsigned)box->z + z, level, + box->x, box->y, box->z, box->width, box->height, + box->depth, dstride, + sgx_resource_face_offset(&r->r, (uint32_t)box->z + z, + level)); + if (r->r.depth && (unsigned)box->z + (unsigned)slices > + r->r.level[level].depth) { + sgx_resource_unmap_level(&r->r); + return; + } + int rows = (int)((box->height + bh - 1u) / bh); + + for (y = 0; y < slices * rows; y++) { + int sl = y / rows, yy = y % rows; + char *row = (char *)p + + (size_t)(r->r.depth ? box->z + sl : 0) * dslice + + (size_t)(box->y / bh + yy) * dstride + + (size_t)(box->x / bw) * bpp; + + memcpy(row, src + (size_t)sl * layer_stride + + (size_t)yy * stride, + (size_t)((box->width + bw - 1u) / bw) * bpp); + /* A format without alpha is stored in one that has + * it, and the caller's rows carry nothing useful in + * that byte. GL says the texture samples alpha as + * one, so it is written here rather than left as + * whatever arrived - an RGB texture blended by its + * source alpha vanishes otherwise. */ + if (r->no_alpha && sgx_format_has_alpha_byte(&r->r)) { + int x; + + for (x = 0; x < (int)box->width; x++) + row[x * 4 + 3] = (char)0xff; + } + } + sgx_resource_unmap_level(&r->r); + } +} + +/* The pattern is the draw module's: its pstipple stage wraps this entry, + * keeps the pattern and uploads it into its own texture. Nothing in the + * hardware takes one, so there is nothing to do here. */ +static void sgx_pipe_set_polygon_stipple(struct pipe_context *pc, + const struct pipe_poly_stipple *st) +{ + (void)pc; (void)st; +} + +static void sgx_pipe_set_clip_state(struct pipe_context *pc, + const struct pipe_clip_state *clip) +{ + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + + draw_set_clip_state(c->draw, clip); +} + +static void sgx_pipe_set_min_samples(struct pipe_context *pc, unsigned n) +{ + (void)pc; (void)n; +} + +/* Submit whatever the frame has accumulated. A caller that is about to look + * at a surface has to see the draws that were meant to be in it. */ +static void sgx_pipe_submit_pending_why(struct sgx_pipe_context *c, + const char *why) +{ + if (!c->ctx.draws && !sgx_clear_pending(&c->ctx)) + return; + { + unsigned n = c->ctx.draws; + int ret; + + int had_clear = sgx_clear_pending(&c->ctx); + + sgx_context_dump_ranges(&c->ctx); + sgx_perf_note(why); + ret = sgx_flush(&c->ctx); + sgx_dbg("readback flush: %u draw(s), %d\n", n, ret); + { + const char *path = SGX_ENVS("SGX_DUMP_RT"); + + if (path) + sgx_dump_target(&c->ctx, path); + } + /* Only when the frame reached the hardware. A flush that + * dropped its frame - sgx_context_drop_frame() forgets the + * pending clear along with everything else - has not painted + * the clear the caller asked for, and turning clearing off + * anyway loses it silently: the next frame keeps the picture + * the last one left, which is the previous draw's, not the + * clear colour. A clear that was never painted stays owed. */ + if (!ret) + sgx_set_clear_enable(&c->ctx, 0); + else if (had_clear) + sgx_request_clear(&c->ctx); + } +} + +static void sgx_pipe_submit_pending(struct sgx_pipe_context *c) +{ + sgx_pipe_submit_pending_why(c, "readback or upload"); +} + +/* ---- occlusion queries ------------------------------------------------- + * + * The ISP counts, per object, into one of eight registers: ISP word B bit 3 + * enables the test and bits [2:0] name the register (EURASIA_ISPB_VISTEST and + * EURASIA_ISPB_VISREG, sgxdefs.h:1435-1438). The registers are registers - + * SGX535 does not define SGX_FEATURE_VISTEST_IN_MEMORY (sgxfeaturedefs.h: + * 189-232) - so the kernel reads them when the render ends and keeps a running + * total per open file, and a query's count is the difference between the total + * when it began and the total when it ended. That is also what makes a query + * survive a frame split: every split fires a render, the kernel sums each one, + * and the difference has all of them. The vendor's microkernel keeps the same + * running total in the client's own buffer and accumulates "after every render + * including SPM partial renders" (3d.asm:1170-1191). + * + * What the count is has one caveat worth stating: the register field is named + * COUNT and is 32 bits (sgx535defs.h:1379-1380), and the unit is a fragment + * that passed the depth and stencil test - which is what the block it lives in + * does - but the DDK in gfx_linux_ddk/ says so nowhere, because it ships no + * query implementation at all. The suite's case is what settles it. + */ + +/* The pool the winsys reads back and the field the ISP word encodes are the + * same eight registers; the two layers name the count separately because + * neither includes the other's header. */ +_Static_assert(SGX_VISTEST_COUNTERS == SGX_ISP_VISTEST_REGS, + "the winsys and the ISP word disagree on the counter count"); + +/* Only what the counter can serve. A predicate is that counter tested + * against zero; this core has no boolean mode of its own - EURASIA_ISPA_VISBOOL + * (sgxdefs.h:1345) belongs to the in-memory form. */ +static int sgx_query_servable(unsigned type) +{ + return type == PIPE_QUERY_OCCLUSION_COUNTER || + type == PIPE_QUERY_OCCLUSION_PREDICATE || + type == PIPE_QUERY_OCCLUSION_PREDICATE_CONSERVATIVE; +} + +static struct pipe_query *sgx_pipe_create_query(struct pipe_context *pc, + unsigned type, unsigned index) +{ + struct sgx_pipe_query *q; + + (void)pc; (void)index; + if (!sgx_query_servable(type)) { + /* By name, so a caller that asked for something this part + * cannot count is told which one rather than left with a + * null it has to guess about. */ + fprintf(sgx_log(), "sgx: %s is not a query this part can " + "answer - the ISP counts fragments that passed, and " + "nothing else\n", util_str_query_type(type, false)); + return NULL; + } + q = CALLOC_STRUCT(sgx_pipe_query); + if (!q) + return NULL; + q->type = type; + q->reg = -1; + return (struct pipe_query *)q; +} + +/* Give the register back. Only once the result is in hand: the count is a + * difference of running totals, so another query taking the register before + * this one has read its end value would add its own draws to it. */ +static void sgx_query_release(struct sgx_pipe_context *c, + struct sgx_pipe_query *q) +{ + if (q->reg < 0) + return; + c->screen->vis_used &= ~(1u << q->reg); + q->reg = -1; +} + +static void sgx_pipe_destroy_query(struct pipe_context *pc, + struct pipe_query *pq) +{ + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + struct sgx_pipe_query *q = (struct sgx_pipe_query *)pq; + + if (!q) + return; + if (c->vis_active == q) { + sgx_set_vistest(&c->ctx, -1); + c->vis_active = NULL; + } + /* A query thrown away without its result may still have a render on + * the core counting into its register. Handing that register on now + * would give the next query a base taken before the harvest, and it + * would count this one's draws as well - so the frame goes and the + * render is drained first. */ + if (q->reg >= 0 && !q->have_result) { + uint64_t counts[SGX_VISTEST_COUNTERS]; + + sgx_pipe_submit_pending_why(c, "occlusion query dropped"); + (void)sgx_vistest_read(c->ctx.ws, 1, counts, NULL); + } + sgx_query_release(c, q); + FREE(q); +} + +static bool sgx_pipe_begin_query(struct pipe_context *pc, struct pipe_query *pq) +{ + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + struct sgx_pipe_query *q = (struct sgx_pipe_query *)pq; + uint64_t counts[SGX_VISTEST_COUNTERS]; + unsigned r; + + if (!q) + return false; + /* ISP word B names one counter per object, so a draw cannot be in two + * queries. Refused rather than served wrong. */ + if (c->vis_active) { + fprintf(sgx_log(), "sgx: a second occlusion query cannot begin " + "while one is counting - the object's ISP word names " + "one counter\n"); + return false; + } + if (q->reg < 0) { + for (r = 0; r < SGX_VISTEST_COUNTERS; r++) + if (!(c->screen->vis_used & (1u << r))) + break; + if (r == SGX_VISTEST_COUNTERS) { + /* The three-bit field cannot name a ninth and two + * queries cannot share one, so the caller is told. + * A register is spoken for from begin_query until its + * result is collected or the query is destroyed. */ + fprintf(sgx_log(), "sgx: all %u visibility counters are " + "in flight - this query cannot begin until one " + "of them is read back\n", + SGX_VISTEST_COUNTERS); + return false; + } + c->screen->vis_used |= 1u << r; + q->reg = (int)r; + } + /* The running total for this register cannot have moved since it was + * free - nothing armed it - so the base needs no drain. */ + if (sgx_vistest_read(c->ctx.ws, 0, counts, NULL)) { + fprintf(sgx_log(), "sgx: the kernel has no visibility counter " + "read-back; occlusion queries need one\n"); + sgx_query_release(c, q); + return false; + } + q->base = counts[q->reg]; + q->result = 0; + q->ended = 0; + q->have_result = 0; + c->vis_active = q; + if (!c->query_paused) + sgx_set_vistest(&c->ctx, q->reg); + return true; +} + +static bool sgx_pipe_end_query(struct pipe_context *pc, struct pipe_query *pq) +{ + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + struct sgx_pipe_query *q = (struct sgx_pipe_query *)pq; + + if (!q || c->vis_active != q) + return false; + sgx_set_vistest(&c->ctx, -1); + c->vis_active = NULL; + q->ended = 1; + /* The counted draws have to reach the core before anything can be + * read back, and only a fired render moves the counters. */ + sgx_pipe_submit_pending_why(c, "occlusion query ended"); + return true; +} + +static bool sgx_pipe_get_query_result(struct pipe_context *pc, + struct pipe_query *pq, bool wait, + union pipe_query_result *result) +{ + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + struct sgx_pipe_query *q = (struct sgx_pipe_query *)pq; + uint64_t counts[SGX_VISTEST_COUNTERS]; + int pending = 0; + + if (!q || !result) + return false; + if (!q->have_result) { + if (!q->ended || q->reg < 0) + return false; + /* A draw made after the query ended may still be sitting in + * the frame, and the frame is what the render is made of. */ + sgx_pipe_submit_pending_why(c, "occlusion query result"); + if (sgx_vistest_read(c->ctx.ws, wait, counts, &pending)) + return false; + /* Without the drain the counts are only what has landed, so a + * render of ours still on the core means not yet. */ + if (!wait && pending) + return false; + q->result = counts[q->reg] - q->base; + q->have_result = 1; + sgx_query_release(c, q); + } + if (q->type == PIPE_QUERY_OCCLUSION_COUNTER) + result->u64 = q->result; + else + result->b = q->result != 0; + return true; +} + +/* The state tracker turns counting off around draws that are not the + * application's - a blit, the mipmap generator - and back on afterwards. The + * query stays open; its records simply stop carrying the counter. */ +static void sgx_pipe_set_active_query_state(struct pipe_context *pc, bool on) +{ + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + + if (c->query_paused == !on) + return; + c->query_paused = !on; + if (c->vis_active) + sgx_set_vistest(&c->ctx, on ? c->vis_active->reg : -1); +} + +static void sgx_pipe_flush_resource(struct pipe_context *pc, + struct pipe_resource *r) +{ + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + + sgx_xfer_flushres++; + (void)r; + sgx_pipe_submit_pending_why(c, "upload: sgx_pipe_flush_resource"); + /* Whoever asked is about to hand this resource to something outside + * this driver, so a blit that has only been promised has to run. */ + sgx_twod_flush(c->ctx.ws); +} + +static void sgx_pipe_set_debug_callback(struct pipe_context *pc, + const struct util_debug_callback *cb) +{ + (void)pc; (void)cb; +} + +/* --- copies ------------------------------------------------------------- + * + * On the CPU. Both resources are linear and mappable, and the state tracker + * uses a blit to upload texture data, so this is on the path of every + * glTexImage2D - but a copy the hardware could do is a copy this driver has no + * frame for yet, and doing it slowly is better than not at all. + */ +/* Levels matter here because Mesa uploads a mip chain one level at a time + * into a staging resource of its own and then copies it across - so a copy + * that ignored the level would stack every level on top of level 0, which is + * exactly what it used to do. */ +static void *sgx_pipe_map_res_level(struct sgx_pipe_context *c, + struct pipe_resource *pres, unsigned level, + unsigned *stride, enum pipe_format *fmt) +{ + struct sgx_pipe_resource *r = (struct sgx_pipe_resource *)pres; + uint32_t ls = 0; + void *p = NULL; + + if (!r) + return NULL; + if (level >= (r->r.nlevels ? r->r.nlevels : 1u)) + level = 0; + if (sgx_resource_map_level(&r->r, c->ctx.ws, level, &p, &ls) != + SGX_RESOURCE_OK) + return NULL; + *stride = ls ? ls : r->r.stride; + *fmt = pres->format; + return p; +} + +static void *sgx_pipe_map_res(struct sgx_pipe_context *c, + struct pipe_resource *pres, unsigned *stride, + enum pipe_format *fmt) +{ + return sgx_pipe_map_res_level(c, pres, 0, stride, fmt); +} + +/* --- the 2D block -------------------------------------------------------- + * + * Copies and solid fills of linear surfaces go to the SGX535's 2D engine: no + * scene, no shader, no tiler, and nothing read through the write-combining + * mapping. It handles the layouts a display server keeps its pixmaps in; + * what it cannot take is refused here and takes the path it always took. + * + * Linear only: the engine walks rows at a byte stride and has no notion of + * the twiddled layout. Same format only: with SRCCOPY the engine moves bits + * and never converts, which is also why a surface the engine has no code for + * can be copied as one with the same pixel size - the code only matters when + * the two ends differ, and here they never do. A fill is the exception: the + * engine converts the ARGB8888 colour word into the destination's code, so a + * fill is only offered for the formats the code really describes. + */ +static uint32_t sgx_twod_code_of(enum sgx_format f, int fill) +{ + switch (f) { + case SGX_FMT_A8R8G8B8: return SGX_2D_FMT_8888ARGB; + case SGX_FMT_R5G6B5: return SGX_2D_FMT_565RGB; + case SGX_FMT_A1R5G5B5: return SGX_2D_FMT_1555ARGB; + case SGX_FMT_A4R4G4B4: return SGX_2D_FMT_4444ARGB; + /* Byte-order and channel-count differences the engine never sees on + * a copy: four bytes are four bytes, two are two, one is one. */ + case SGX_FMT_A8B8G8R8: return fill ? 0 : SGX_2D_FMT_8888ARGB; + case SGX_FMT_AL88: + case SGX_FMT_YUY2: + case SGX_FMT_UYVY: return fill ? 0 : SGX_2D_FMT_565RGB; + case SGX_FMT_A8: return fill ? 0 : SGX_2D_FMT_332RGB; + default: return 0; + } +} + +/* A level of a resource as the engine sees it, or 0 for one it cannot. */ +static int sgx_twod_surf_of(struct pipe_resource *pres, unsigned level, + int fill, struct sgx_twod_surf *out, + unsigned *w, unsigned *h) +{ + struct sgx_pipe_resource *r = (struct sgx_pipe_resource *)pres; + uint32_t code; + + if (!r || pres->target == PIPE_BUFFER || r->r.twiddled || + !r->r.bound || !r->r.bo.gpu_va || r->r.bo.window != SGX_VM_SURFACE) + return 0; + code = sgx_twod_code_of(r->r.format, fill); + if (!code) + return 0; + if (r->r.nlevels) { + if (level >= r->r.nlevels) + return 0; + out->offset = r->r.level[level].offset; + out->stride = r->r.level[level].stride; + *w = r->r.level[level].width; + *h = r->r.level[level].height; + } else { + if (level) + return 0; + out->offset = 0; + out->stride = r->r.stride; + *w = r->r.width; + *h = r->r.height; + } + if (!out->stride || (out->stride & 3u) || (out->offset & 3u)) + return 0; + out->bo = &r->r.bo; + out->format = code; + return 1; +} + +/* Below this many bytes a copy stays on the CPU. + * + * It was 4096, because a job was one synchronous ioctl of about 48 us against + * the engine's 420 MB/s and the CPU's 36 MB/s over the write-combining + * mapping - so a glyph, a few hundred bytes, measured 25x slower on the + * engine. A copy no longer costs an ioctl of its own: it is appended to the + * job the winsys keeps open, and one ioctl carries the whole string. What is + * left is the engine's own per-blit cost against a short memcpy, which is why + * there is still a floor and why it is small. SGX_2D_MIN overrides, in bytes; + * 0 puts everything on the engine. */ +static uint64_t sgx_twod_min_bytes(void) +{ + static uint64_t min; + static int set; + + if (!set) { + const char *e = getenv("SGX_2D_MIN"); + + min = e && *e ? strtoull(e, NULL, 0) : 256u; + set = 1; + } + return min; +} + +/* Nonzero once the copy is done on the engine. Zero means the caller takes + * its usual path; a job the kernel refused or lost also lands there, since + * the CPU copy that follows writes the whole rectangle again. */ +static int sgx_pipe_twod_copy(struct sgx_pipe_context *c, + struct pipe_resource *dst, unsigned dst_level, + unsigned dstx, unsigned dsty, + struct pipe_resource *src, unsigned src_level, + const struct pipe_box *box) +{ + struct sgx_twod_surf d, s; + unsigned dw, dh, sw, sh; + int ret; + + /* Everything that can be decided from the arguments comes first, and + * nothing above the last of them may have an effect the caller would + * notice - a copy the engine will not take has to cost what it cost + * before the engine existed, which is nothing. sgx_twod_available() + * is below them because the first call to it asks the kernel. */ + if (dst->format != src->format || box->depth != 1 || box->z || + box->width <= 0 || box->height <= 0 || box->x < 0 || box->y < 0) + return sgx_twod_no(SGX_TWOD_FORMAT); + if (!sgx_twod_surf_of(dst, dst_level, 0, &d, &dw, &dh) || + !sgx_twod_surf_of(src, src_level, 0, &s, &sw, &sh)) + return sgx_twod_no(SGX_TWOD_SURF); + if (dstx + box->width > dw || dsty + box->height > dh || + (unsigned)box->x + box->width > sw || + (unsigned)box->y + box->height > sh) + return sgx_twod_no(SGX_TWOD_BOUNDS); + if ((uint64_t)box->width * box->height * + util_format_get_blocksize(dst->format) < sgx_twod_min_bytes()) + return sgx_twod_no(SGX_TWOD_SMALL); + if (SGX_ENVS("SGX_NO_2D") || !sgx_twod_available(c->ctx.ws)) + return sgx_twod_no(SGX_TWOD_OFF); + /* The draws accumulated so far are meant to have happened before this + * copy, and the kernel drains the render they become before it runs + * the job - so nothing here waits on its own. */ + sgx_pipe_submit_pending_why(c, "upload: sgx_pipe_twod_copy"); + /* Appended, not run: the winsys hands the job over when something has + * to see it. Refused here means the caller takes its usual path, so + * everything that could refuse the job is tested at this point - once + * it is in the job there is no falling back. */ + ret = sgx_twod_queue_copy(c->ctx.ws, &d, (int)dstx, (int)dsty, &s, + box->x, box->y, box->width, box->height); + if (ret) { + sgx_dbg("2D copy not offered: %d\n", ret); + return sgx_twod_no(SGX_TWOD_REFUSED); + } + sgx_twod_why[SGX_TWOD_TOOK]++; + /* Kept open only when the caller has asked for it. The queue is what + * makes a job cost one ioctl for a whole string of copies, and it is + * worth having - but a queued copy is the same unverifiable promise + * the fill above refuses to make, and two rounds of blank rectangles + * came of promises. SGX_2D_BATCH=1 opens the job; the default runs + * each copy before returning, which is what the caller believes + * happened. */ + if (!SGX_ENVS("SGX_2D_BATCH")) { + ret = sgx_twod_flush(c->ctx.ws); + if (ret) { + fprintf(sgx_log(), "sgx: a 2D copy of %dx%d was " + "refused (%d); copying on the CPU instead\n", + box->width, box->height, ret); + return sgx_twod_no(SGX_TWOD_REFUSED); + } + } + sgx_dbg("2D copy: %dx%d from %d,%d to %u,%u, level %u->%u %s\n", + box->width, box->height, box->x, box->y, dstx, dsty, src_level, + dst_level, SGX_ENVS("SGX_2D_BATCH") ? "queued" : "done"); + return 1; +} + +/* The engine takes its colour as ARGB8888 and converts to the destination's + * code; a surface stored the other way round has to be told so. */ +static uint32_t sgx_twod_colour(enum sgx_format f, + const union pipe_color_union *color) +{ + uint32_t r = (uint32_t)(CLAMP(color->f[0], 0.0f, 1.0f) * 255.0f + 0.5f); + uint32_t g = (uint32_t)(CLAMP(color->f[1], 0.0f, 1.0f) * 255.0f + 0.5f); + uint32_t b = (uint32_t)(CLAMP(color->f[2], 0.0f, 1.0f) * 255.0f + 0.5f); + uint32_t a = (uint32_t)(CLAMP(color->f[3], 0.0f, 1.0f) * 255.0f + 0.5f); + + (void)f; + return a << 24 | r << 16 | g << 8 | b; +} + +static int sgx_pipe_twod_fill(struct sgx_pipe_context *c, + struct pipe_surface *dst, + const union pipe_color_union *color, + unsigned x, unsigned y, unsigned w, unsigned h) +{ + struct sgx_pipe_resource *r = (struct sgx_pipe_resource *)dst->texture; + struct sgx_twod_surf d; + unsigned dw, dh; + int ret; + + /* The arguments first, the kernel last: see sgx_pipe_twod_copy(). */ + if (!w || !h) + return 0; + if (!sgx_twod_surf_of(dst->texture, dst->level, 1, &d, &dw, &dh)) + return 0; + if (x + w > dw || y + h > dh || dst->first_layer) + return 0; + if (SGX_ENVS("SGX_NO_2D") || !sgx_twod_available(c->ctx.ws)) + return 0; + /* The draws the caller has accumulated are meant to have happened + * before this clear, so they go first - on this path and on the CPU + * one below it, which makes the same call for the same reason. It is + * not the engine's requirement: it is what "clear this surface now" + * means when there is a frame open on the same surface, and leaving + * it out puts the fill under the drawing it was meant to precede. + * That is how twm's menu lost its text once already. + * + * Below the refusals, so a clear the engine will not take does not + * pay for it here - it pays for it once, in the CPU path. */ + sgx_pipe_submit_pending_why(c, "upload: sgx_pipe_twod_fill"); + ret = sgx_twod_queue_fill(c->ctx.ws, &d, (int)x, (int)y, (int)w, + (int)h, sgx_twod_colour(r->r.format, color)); + if (ret) { + sgx_dbg("2D fill not offered: %d\n", ret); + return 0; + } + /* Run it now, and say it was done only if it was. + * + * A queued blit is a promise, and the caller of a clear has no way to + * find out later that the promise was broken: it returns believing + * the surface is cleared, and if the job is refused - or simply never + * closed - the surface keeps whatever it held, which for a fresh + * pixmap is black. That is a promise the caller cannot check, so it + * is not made. The fallback below is still available at this instant, + * because nothing has moved between the queue and here; a moment + * later it would not be. + * + * Clears are rare next to copies - one in several hundred flushes on + * a display server - so the ioctl this costs is not the cost the open + * job was opened to avoid. */ + ret = sgx_twod_flush(c->ctx.ws); + if (ret) { + fprintf(sgx_log(), "sgx: a 2D fill of %ux%u was refused (%d); " + "clearing on the CPU instead\n", w, h, ret); + return 0; + } + sgx_dbg("2D fill: %ux%u at %u,%u done\n", w, h, x, y); + return 1; +} + +/* pipe_context::clear_render_target. The engine when it can; otherwise the + * CPU through the level's mapping, the way the copy below does it. Mesa's + * util_clear_render_target() would do the same map-and-fill, but through a + * transfer of the whole surface. */ +static void sgx_pipe_clear_render_target(struct pipe_context *pc, + struct pipe_surface *dst, + const union pipe_color_union *color, + unsigned dstx, unsigned dsty, + unsigned width, unsigned height, + bool render_condition_enabled) +{ + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + struct sgx_pipe_resource *r = (struct sgx_pipe_resource *)dst->texture; + unsigned stride = 0; + enum pipe_format fmt = PIPE_FORMAT_NONE; + void *p; + + (void)render_condition_enabled; + if (!r || !width || !height) + return; + if (sgx_pipe_twod_fill(c, dst, color, dstx, dsty, width, height)) + return; + sgx_pipe_submit_pending_why(c, "upload: sgx_pipe_clear_render_target"); + sgx_wait_idle(c->ctx.ws); + p = sgx_pipe_map_res_level(c, dst->texture, dst->level, &stride, &fmt); + if (!p) { + /* Loud. Mesa has no fallback behind this hook - before the + * driver had one at all, the state tracker cleared with a + * draw of its own - so a quiet return here is a surface the + * caller believes cleared and nothing cleared. */ + fprintf(sgx_log(), "sgx: clear_render_target: the surface " + "would not map; it has NOT been cleared\n"); + return; + } + if (dst->level < (r->r.nlevels ? r->r.nlevels : 1u)) { + unsigned lw = r->r.nlevels ? r->r.level[dst->level].width : r->r.width; + unsigned lh = r->r.nlevels ? r->r.level[dst->level].height : r->r.height; + + if (dstx < lw && dsty < lh) { + union util_color uc; + + if (width > lw - dstx) + width = lw - dstx; + if (height > lh - dsty) + height = lh - dsty; + util_pack_color(color->f, dst->format, &uc); + util_fill_rect(p, dst->format, stride, dstx, dsty, width, + height, &uc); + if (SGX_ENVS("SGX_CLEAR_TRACE")) + fprintf(sgx_log(), "sgx: cpu clear: %ux%u at " + "%u,%u level %ux%u stride %u rgba " + "%.3f %.3f %.3f %.3f\n", width, height, + dstx, dsty, lw, lh, stride, + color->f[0], color->f[1], color->f[2], + color->f[3]); + } else { + fprintf(sgx_log(), "sgx: clear_render_target: %u,%u is " + "outside the %ux%u level; it has NOT been " + "cleared\n", dstx, dsty, lw, lh); + } + } else { + fprintf(sgx_log(), "sgx: clear_render_target: level %u of %u " + "does not exist; it has NOT been cleared\n", + dst->level, r->r.nlevels); + } + sgx_resource_unmap_level(&r->r); +} + +/* What the caller had bound, handed to the blitter so it can put it back + * after binding its own. The blitter restores from these; the driver keeps + * them for no other reason. */ +static void sgx_blitter_save(struct sgx_pipe_context *c) +{ + util_blitter_save_blend(c->blitter, c->cur_blend); + util_blitter_save_depth_stencil_alpha(c->blitter, c->cur_dsa); + util_blitter_save_stencil_ref(c->blitter, &c->stencil_ref); + util_blitter_save_rasterizer(c->blitter, c->cur_rast); + util_blitter_save_fragment_shader(c->blitter, c->cur_fs); + util_blitter_save_vertex_shader(c->blitter, c->vs); + util_blitter_save_vertex_elements(c->blitter, c->velems); + util_blitter_save_viewport(c->blitter, &c->viewport); + util_blitter_save_scissor(c->blitter, &c->scissor_state); + util_blitter_save_framebuffer(c->blitter, &c->fb); + util_blitter_save_fragment_sampler_states(c->blitter, + c->ncur_samplers, + c->cur_samplers); + util_blitter_save_fragment_sampler_views(c->blitter, c->ncur_views, + c->cur_views); + util_blitter_save_vertex_buffers(c->blitter, c->vb, c->nvb); + util_blitter_save_fragment_constant_buffer_slot(c->blitter, + &c->cur_fs_cb); +} + +static void sgx_pipe_resource_copy_region(struct pipe_context *pc, + struct pipe_resource *dst, + unsigned dst_level, unsigned dstx, + unsigned dsty, unsigned dstz, + struct pipe_resource *src, + unsigned src_level, + const struct pipe_box *box) +{ + /* The copy writes the object behind dst, so any cached copy of it is + * out of date. */ + sgx_pipe_shadow_dirty(dst); + sgx_xfer_copy++; + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + unsigned ds = 0, ss = 0; + enum pipe_format df = PIPE_FORMAT_NONE, sf = PIPE_FORMAT_NONE; + void *dp, *sp; + + (void)dstz; + /* The 2D block first: a copy it can take costs no scene, no shader + * and no tiler, and it never disturbs the frame the caller has open. */ + if (!dstz && sgx_pipe_twod_copy(c, dst, dst_level, dstx, dsty, src, + src_level, box)) + return; + /* On the part, as a textured draw, whenever it can be. The path below + * is a CPU copy, and the CPU reads this memory through a + * write-combining mapping - which is an order of magnitude slower than + * a cached read, and is what made a window move take seconds: 86% of + * the server's time went to util_copy_rect while the core sat idle. + * + * Linear surfaces of one format only. A twiddled level has no row + * order - the CPU path goes through a map that knows the swizzle - and + * a copy that also converts is not a copy: taking those broke every + * texture test, which is what a texture upload is, a 32x32 from one + * format into another. util_blitter_is_copy_supported() accepts both; + * the hardware here does not. What is left is the case that matters, + * a surface copied to a surface like it, which is every window move + * and every scroll. + * + * in_blit stops the recursion: the blitter draws, the draw flushes, + * and a flush can want a copy of its own. */ + if (c->blitter && !c->in_blit && !SGX_ENVS("SGX_NO_GPU_COPY") && + dst->target != PIPE_BUFFER && src->target != PIPE_BUFFER && + !((struct sgx_pipe_resource *)dst)->r.twiddled && + !((struct sgx_pipe_resource *)src)->r.twiddled && + dst->format == src->format && + util_blitter_is_copy_supported(c->blitter, dst, src)) { + /* The caller's frame goes first. A frame carries one render + * target and one set of state, and the blit is a draw into a + * different target - joining what the caller has open mixes + * the two, which took out even the shader cases that sample + * nothing. */ + sgx_dbg("gpu copy: %ux%u+%d+%d -> +%u+%u, dst fmt %u tw %u, " + "src fmt %u tw %u\n", box->width, box->height, box->x, + box->y, dstx, dsty, (unsigned)dst->format, + ((struct sgx_pipe_resource *)dst)->r.twiddled, + (unsigned)src->format, + ((struct sgx_pipe_resource *)src)->r.twiddled); + c->in_blit = 1; + sgx_pipe_submit_pending_why(c, "upload: sgx_pipe_resource_copy_region"); + sgx_blitter_save(c); + util_blitter_copy_texture(c->blitter, dst, dst_level, dstx, + dsty, dstz, src, src_level, box); + c->in_blit = 0; + return; + } + /* This copy is done by the CPU, so the frame the caller has been + * building has to have happened first. sgx_resource_map() waits for + * what is already submitted, which is not the same thing: draws sit + * accumulated until something flushes them, and reading the source + * before that copies the surface as it was without them. + * + * That is how twm's menu lost its text. The X server rendered the + * menu into its own pixmap - correctly, the pixmap dumps show every + * glyph - and then copied it to the screen with a CopyArea, which + * lands here. The glyphs were still pending, so what reached the + * screen was the background fill that had been flushed earlier. */ + sgx_pipe_submit_pending_why(c, "upload: sgx_pipe_resource_copy_region"); + sgx_wait_idle(c->ctx.ws); + dp = sgx_pipe_map_res_level(c, dst, dst_level, &ds, &df); + sp = sgx_pipe_map_res_level(c, src, src_level, &ss, &sf); + sgx_dbg("copy: level %u->%u, %dx%d from %d,%d to %u,%u, stride %u->%u, " + "%s res\n", src_level, dst_level, box->width, box->height, + box->x, box->y, dstx, dsty, ss, ds, + src == dst ? "same" : "two"); + if (!dp || !sp) { + sgx_dbg("copy: a resource would not map\n"); + goto out; + } + if (dst->target == PIPE_BUFFER || src->target == PIPE_BUFFER) { + unsigned bytes = box->width; + + /* A buffer's width0 is its size in bytes, and box->width here + * is a byte count the caller chose. Unchecked, this overwrites + * the heap past either mapping. */ + if ((size_t)dstx + bytes > dst->width0 || + (size_t)box->x + bytes > src->width0) { + sgx_dbg("copy: %u byte(s) at %u/%d is past the %u/%u " + "the buffers hold\n", bytes, dstx, box->x, + dst->width0, src->width0); + goto out; + } + memcpy((char *)dp + dstx, (const char *)sp + box->x, bytes); + goto out; + } + if (!util_format_translate(df, dp, ds, dstx, dsty, + sf, sp, ss, box->x, box->y, + box->width, box->height)) + sgx_dbg("copy: %s to %s is not a conversion util_format has\n", + util_format_short_name(sf), util_format_short_name(df)); + /* What landed, read back out of the destination: a copy that names the + * right rectangle says nothing about the bytes in it. */ + if (SGX_ENVS("SGX_DUMP_COPY") && box->height > 4) { + const uint32_t *row = (const uint32_t *)((const char *)dp + + (size_t)(dsty + box->height / 2) * ds) + + dstx; + unsigned q; + + fprintf(sgx_log(), "sgx: copy %s->%s %dx%d to %u,%u mid row:", + util_format_short_name(sf), util_format_short_name(df), + box->width, box->height, dstx, dsty); + for (q = 0; q < (unsigned)box->width && q < 8; q++) + fprintf(sgx_log(), " %08x", row[q]); + fprintf(sgx_log(), "\n"); + } +out: + /* The twiddle happens on unmap, so the destination has to be released + * through the level-aware path or the copy never reaches the object. */ + if (sp) + sgx_resource_unmap_level(&((struct sgx_pipe_resource *)src)->r); + if (dp) + sgx_resource_unmap_level(&((struct sgx_pipe_resource *)dst)->r); +} + +static void sgx_pipe_blit(struct pipe_context *pc, + const struct pipe_blit_info *info) +{ + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + unsigned ds = 0, ss = 0; + enum pipe_format df = PIPE_FORMAT_NONE, sf = PIPE_FORMAT_NONE; + void *dp, *sp; + int x, y; + + if (info->src.box.width == info->dst.box.width && + info->src.box.height == info->dst.box.height) { + sgx_pipe_resource_copy_region(pc, info->dst.resource, + info->dst.level, info->dst.box.x, + info->dst.box.y, info->dst.box.z, + info->src.resource, + info->src.level, &info->src.box); + return; + } + /* Scaled, so each destination pixel is sampled rather than copied. + * Nearest only, which is what glGenerateMipmap gets: a box filter + * would be the better reduction, but the state tracker only asks for + * this when the caller declined to supply the level itself. */ + dp = sgx_pipe_map_res_level(c, info->dst.resource, info->dst.level, + &ds, &df); + sp = sgx_pipe_map_res_level(c, info->src.resource, info->src.level, + &ss, &sf); + if (!dp || !sp) { + sgx_dbg("blit: a resource would not map\n"); + goto out; + } + for (y = 0; y < info->dst.box.height; y++) { + for (x = 0; x < info->dst.box.width; x++) { + int sx = info->src.box.x + x * info->src.box.width / + info->dst.box.width; + int sy = info->src.box.y + y * info->src.box.height / + info->dst.box.height; + + util_format_translate(df, dp, ds, + info->dst.box.x + x, + info->dst.box.y + y, + sf, sp, ss, sx, sy, 1, 1); + } + } +out: + if (sp) + sgx_resource_unmap_level(&((struct sgx_pipe_resource *)info->src.resource)->r); + if (dp) + sgx_resource_unmap_level(&((struct sgx_pipe_resource *)info->dst.resource)->r); +} + +/* The channels of a texel, for the reduction below: where each sits in the + * little-endian texel and how wide it is. Every channel is averaged on its + * own, which is what a box filter is; the eight-bit formats are one byte a + * channel and the sixteen-bit ones are unpacked from their fields. Zero + * channels is a format the filter does not know, which the caller refuses. */ +struct sgx_channel { unsigned char shift, bits; }; + +static unsigned sgx_texel_channels(enum sgx_format fmt, + struct sgx_channel ch[4]) +{ + static const struct sgx_channel c8[4] = { + { 0, 8 }, { 8, 8 }, { 16, 8 }, { 24, 8 } }; + static const struct sgx_channel c565[3] = { + { 0, 5 }, { 5, 6 }, { 11, 5 } }; + static const struct sgx_channel c1555[4] = { + { 0, 5 }, { 5, 5 }, { 10, 5 }, { 15, 1 } }; + static const struct sgx_channel c4444[4] = { + { 0, 4 }, { 4, 4 }, { 8, 4 }, { 12, 4 } }; + unsigned n; + + switch (fmt) { + case SGX_FMT_A8: n = 1; memcpy(ch, c8, sizeof *ch * n); return n; + case SGX_FMT_AL88: n = 2; memcpy(ch, c8, sizeof *ch * n); return n; + case SGX_FMT_A8R8G8B8: + case SGX_FMT_A8B8G8R8: n = 4; memcpy(ch, c8, sizeof *ch * n); return n; + case SGX_FMT_R5G6B5: n = 3; memcpy(ch, c565, sizeof c565); return n; + case SGX_FMT_A1R5G5B5: n = 4; memcpy(ch, c1555, sizeof c1555); return n; + case SGX_FMT_A4R4G4B4: n = 4; memcpy(ch, c4444, sizeof c4444); return n; + default: return 0; + } +} + +static uint32_t sgx_texel_load(const unsigned char *p, unsigned bpp) +{ + uint32_t v = 0; + unsigned i; + + for (i = 0; i < bpp; i++) + v |= (uint32_t)p[i] << (8u * i); + return v; +} + +static void sgx_texel_store(unsigned char *p, unsigned bpp, uint32_t v) +{ + unsigned i; + + for (i = 0; i < bpp; i++) + p[i] = (unsigned char)(v >> (8u * i)); +} + +/* One level from the one above it: each texel the rounded mean of the 2x2 + * block it covers, which is the box filter glGenerateMipmap is specified to + * produce and the vendor's own CPU path (opengles2/makemips.c:244-250, + * (sum + 2) >> 2). A source side that is not twice the destination's - + * an odd size, or a side already at one - clamps the block to what exists. */ +static void sgx_reduce_level(unsigned char *dst, uint32_t dstride, + unsigned dw, unsigned dh, + const unsigned char *src, uint32_t sstride, + unsigned sw, unsigned sh, unsigned bpp, + const struct sgx_channel *ch, unsigned nch) +{ + unsigned x, y, k; + + for (y = 0; y < dh; y++) { + for (x = 0; x < dw; x++) { + unsigned x0 = x * 2 < sw ? x * 2 : sw - 1; + unsigned y0 = y * 2 < sh ? y * 2 : sh - 1; + unsigned x1 = x0 + 1 < sw ? x0 + 1 : x0; + unsigned y1 = y0 + 1 < sh ? y0 + 1 : y0; + uint32_t t[4], o = 0; + + t[0] = sgx_texel_load(src + (size_t)y0 * sstride + + (size_t)x0 * bpp, bpp); + t[1] = sgx_texel_load(src + (size_t)y0 * sstride + + (size_t)x1 * bpp, bpp); + t[2] = sgx_texel_load(src + (size_t)y1 * sstride + + (size_t)x0 * bpp, bpp); + t[3] = sgx_texel_load(src + (size_t)y1 * sstride + + (size_t)x1 * bpp, bpp); + for (k = 0; k < nch; k++) { + uint32_t m = (1u << ch[k].bits) - 1u, v = 0; + unsigned i; + + for (i = 0; i < 4; i++) + v += (t[i] >> ch[k].shift) & m; + o |= ((v + 2u) / 4u) << ch[k].shift; + } + sgx_texel_store(dst + (size_t)y * dstride + + (size_t)x * bpp, bpp, o); + } + } +} + +/* Mesa asks the driver first and only falls back to its own path if this + * returns false - and its fallback wants to render into each level, which this + * driver cannot do for a twiddled surface. The vendor reduces on the transfer + * queue where it can and on the CPU where it cannot (opengles2/makemips.c: + * HardwareMakeTextureMipmapLevels, MakeMapLevel8/16/32bpp); this driver has no + * transfer queue, and the CPU path is what glGenerateMipmap is specified to + * produce. Every face of a cube, since each is a chain of its own. */ +static bool sgx_pipe_generate_mipmap(struct pipe_context *pc, + struct pipe_resource *pres, + enum pipe_format format, + unsigned base_level, unsigned last_level, + unsigned first_layer, unsigned last_layer) +{ + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + struct sgx_pipe_resource *r = (struct sgx_pipe_resource *)pres; + unsigned bpp = sgx_format_bytes(r ? r->r.format : SGX_FMT_NONE); + struct sgx_channel ch[4]; + unsigned nch, l, f, faces; + + (void)format; + if (!r || !bpp || !r->r.nlevels || last_level >= r->r.nlevels) + return false; + /* A format the filter cannot take apart is refused so Mesa does it, + * and so is a volume: the reduction here is a 2D one. */ + nch = sgx_texel_channels(r->r.format, ch); + if (!nch || r->r.depth) + return false; + faces = r->r.nfaces ? r->r.nfaces : 1u; + if (first_layer >= faces || last_layer >= faces || + last_layer < first_layer) + return false; + + for (f = first_layer; f <= last_layer; f++) { + for (l = base_level + 1; l <= last_level; l++) { + unsigned char *src = NULL, *dst = NULL, *keep; + uint32_t sstride = 0, dstride = 0; + unsigned sw, sh; + + /* One level at a time, and the source unmapped + * before the destination is mapped: a twiddled + * resource has one staging buffer, so holding both + * at once would alias them. */ + if (sgx_resource_map_face_level(&r->r, c->ctx.ws, f, + l - 1u, (void **)&src, + &sstride) != + SGX_RESOURCE_OK) + return false; + sw = r->r.level[l - 1u].width; + sh = r->r.level[l - 1u].height; + if (!sstride) + sstride = bpp * sw; + keep = malloc((size_t)sstride * sh); + if (!keep) { + sgx_resource_unmap_level(&r->r); + return false; + } + memcpy(keep, src, (size_t)sstride * sh); + sgx_resource_unmap_level(&r->r); + if (sgx_resource_map_face_level(&r->r, c->ctx.ws, f, l, + (void **)&dst, + &dstride) != + SGX_RESOURCE_OK) { + free(keep); + return false; + } + if (!dstride) + dstride = bpp * r->r.level[l].width; + sgx_reduce_level(dst, dstride, r->r.level[l].width, + r->r.level[l].height, keep, sstride, + sw, sh, bpp, ch, nch); + sgx_resource_unmap_level(&r->r); + free(keep); + } + } + sgx_dbg("genmip: levels %u..%u of %ux%u, faces %u..%u\n", + base_level + 1u, last_level, r->r.width, r->r.height, + first_layer, last_layer); + return true; +} + +/* "Flush any pending framebuffer writes and invalidate texture caches", which + * on this part is the frame boundary and nothing else. + * + * A scene's writes reach memory when its render ends, and the kernel fires one + * render at a time, so the scene that samples the result cannot start before + * the one that wrote it finished. The caches are invalidated by the kick + * itself - sgx_kick_preamble() flushes the bus interface and hands the USSE a + * cache invalidate before every scene - so there is nothing further to do than + * end the frame. + * + * It was a no-op while the cap was off, which is why nothing depended on it. */ +static void sgx_pipe_texture_barrier(struct pipe_context *pc, unsigned flags) +{ + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + + (void)flags; + if (!c || !c->ctx.draws) + return; + sgx_perf_note("texture barrier"); + if (!sgx_flush(&c->ctx)) + sgx_set_clear_enable(&c->ctx, 0); +} + +static void sgx_pipe_memory_barrier(struct pipe_context *pc, unsigned flags) +{ + (void)pc; (void)flags; +} + +static void sgx_pipe_flush(struct pipe_context *pc, + struct pipe_fence_handle **fence, unsigned flags) +{ + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + + (void)flags; + c->draws_seen = 0; + /* A clear with no draw behind it is still work. Testing only for + * draws here is what made glClear do nothing at all: the frame's + * first record is the clear itself, and sgx_flush() knows that, but + * it was never reached. */ + if (c->ctx.draws || sgx_clear_pending(&c->ctx)) { + /* Read before the flush: it zeroes the counter on success, so + * reporting afterwards always said none. */ + unsigned n = c->ctx.draws; + int had_clear = sgx_clear_pending(&c->ctx); + int ret; + + /* Dumped before the flush for the same reason the count is + * read before it: the flush resets the records, so afterwards + * there is nothing left to report. */ + sgx_context_dump_ranges(&c->ctx); + sgx_perf_note("caller asked to flush"); + ret = sgx_flush(&c->ctx); + + sgx_dbg("flush: %u draw(s), %u dropped, %u primitive(s) with " + "no path, %d\n", n, sgx_context_dropped(&c->ctx), + c->ctx.prim_dropped, ret); + { + const uint32_t *h = c->ctx.bo[SGX_CTX_BO_HEAP].map; + + sgx_dbg("flush: tex0 va 0x%llx size %llu bound %d, " + "texstate %08x %08x ctl %08x\n", + (unsigned long long) + c->ctx.bo[SGX_CTX_BO_TEX].gpu_va, + (unsigned long long) + c->ctx.bo[SGX_CTX_BO_TEX].size, + c->ctx.bound[SGX_CTX_BO_TEX], + h ? h[XPSB_TEXSTATE_OFF / 4] : 0, + h ? h[XPSB_TEXSTATE_OFF / 4 + 1] : 0, + h ? h[XPSB_TEXCTL_OFF / 4] : 0); + /* Gated like everything around it. These two were bare + * fprintf()s, so every flush wrote a line and up to + * thirty-two hex words to stderr whether or not + * anything was being debugged - per frame, on a part + * whose frames are measured in tens of milliseconds, + * and into the X server's log for every glamor + * operation. */ + if (h && sgx_debug()) { + const uint32_t *b = h + SGX_FS_CONST_OFF / 4; + unsigned q; + + unsigned nq = c->ctx.fs ? + c->ctx.fs->pool_base + + c->ctx.fs->pool_dwords : 12u; + + if (nq < 12u) + nq = 12u; + if (nq > 32u) + nq = 32u; + fprintf(sgx_log(), + "sgx: flush: vs %u insn %u temp %u secattr %u limm, hwtcl %d, sa %u, vtx_floats %u\n", + c->ctx.vs ? c->ctx.vs->ninsns : 0, + c->ctx.vs ? c->ctx.vs->ntemps : 0, + c->ctx.vs ? c->ctx.vs->nsecattr : 0, + c->ctx.vs ? c->ctx.vs->nlimm : 0, + c->hwtcl, c->ctx.sa_dwords, + c->ctx.vtx_floats); + fprintf(sgx_log(), "sgx: flush: sa bank"); + for (q = 0; q < nq; q++) + fprintf(sgx_log(), " %s%08x", + q % 4 ? "" : "| ", b[q]); + fprintf(sgx_log(), " (preiter %d, nuni %u, " + "pool_base %u)\n", + c->ctx.fs ? + (int)c->ctx.fs->tex_preiterated : -1, + c->ctx.fs ? c->ctx.fs->nuniform : 0, + c->ctx.fs ? c->ctx.fs->pool_base : 0); + } + } + /* A whole picture, not the one pixel a test happens to look + * at: SGX_DUMP_RT names a file the colour buffer is written + * to after each frame, as a binary PPM. Reasoning from a + * single sampled pixel is how several evenings went. */ + { + const char *path = SGX_ENVS("SGX_DUMP_RT"); + /* SGX_DUMP_RT_W dumps only frames whose target is that + * wide, each to its own file: under X every frame + * overwrites the last, so a pixmap drawn once is gone + * before it can be looked at. */ + const char *wo = SGX_ENVS("SGX_DUMP_RT_W"); + + if (path && wo && *wo) { + static unsigned seq; + + if (c->ctx.fb.width == (unsigned)atoi(wo)) { + char nm[256]; + + snprintf(nm, sizeof nm, "%s-%u.ppm", + path, seq++); + sgx_dump_target(&c->ctx, nm); + } + } else if (path) { + sgx_dump_target(&c->ctx, path); + } + } + /* A frame that did not reach the hardware is a frame the + * caller will never see, and the return was read only to be + * logged under SGX_DEBUG. Said once: a submit that fails + * usually keeps failing, and a message per frame buries the + * first one. */ + if (ret) { + static int said; + + if (!said) { + said = 1; + fprintf(sgx_log(), "sgx: a frame was not " + "submitted (%d) - what it drew will " + "not appear\n", ret); + } + } + /* The next frame keeps what this one drew unless its caller + * asks for a clear of its own - but only if this one drew. + * A frame that was dropped painted no clear, and forgetting + * it here leaves the previous picture standing. */ + if (!ret) + sgx_set_clear_enable(&c->ctx, 0); + else if (had_clear) + sgx_request_clear(&c->ctx); + } else { + sgx_dbg("flush: nothing to submit\n"); + } + /* After the frame, because the frame path flushes the job itself and + * the order is blits first; this catches a flush with no draws behind + * it, which is what a display server's copy-only frame is. The fence + * below is handed out signalled, so it must not be handed out with a + * blit still promised. */ + sgx_twod_flush(c->ctx.ws); + if (fence) + *fence = (struct pipe_fence_handle *)&sgx_the_fence; + if (SGX_ENVS("SGX_XFER_STATS")) { + static unsigned n; + + if (!(++n % 32u)) { + unsigned q; + + fprintf(sgx_log(), "sgx: xfer over %u flush(es): " + "texmap %u bufmap %u texsub %u bufsub %u " + "copy %u flushres %u\n", n, sgx_xfer_texmap, + sgx_xfer_bufmap, sgx_xfer_texsub, + sgx_xfer_bufsub, sgx_xfer_copy, + sgx_xfer_flushres); + fprintf(sgx_log(), "sgx: 2D copies:"); + for (q = 0; q < SGX_TWOD_WHY_N; q++) + fprintf(sgx_log(), " %s %u", + sgx_twod_why_name[q], sgx_twod_why[q]); + fprintf(sgx_log(), "\n"); + } + } + sgx_perf_swap(); +} + +/* A short if is cheaper flattened into a select than branched around, so the + * small ones still are. The limit used to be infinite because the backend had + * no conditional jump at all, and that cost more than it saved: flattening + * keeps both arms of every if live at once, and the X server's composite + * shader came to fifty-four temporaries where the task's count can name + * thirty-one, so it was refused and nothing was drawn. With TEST decoded it + * branches instead and comes to thirty. + * + * SGX_CF_LIMIT sets the limit, 0 branching everything, for comparing the two. + */ +#define SGX_CF_FLATTEN_LIMIT 8u +static void sgx_nir_flatten_cf(struct nir_shader *nir) +{ + nir_opt_peephole_select_options flatten = { + .limit = SGX_CF_FLATTEN_LIMIT, + .indirect_load_ok = true, + .expensive_alu_ok = true, + .discard_ok = true, + }; + const char *e = SGX_ENVS("SGX_CF_LIMIT"); + bool progress; + unsigned round = 0; + + if (e && *e) + flatten.limit = (unsigned)atoi(e); + + do { + progress = false; + NIR_PASS(progress, nir, nir_opt_loop); + NIR_PASS(progress, nir, nir_opt_loop_unroll); + NIR_PASS(progress, nir, nir_opt_if, 0); + NIR_PASS(progress, nir, nir_opt_peephole_select, &flatten); + NIR_PASS(progress, nir, nir_opt_copy_prop); + NIR_PASS(progress, nir, nir_opt_constant_folding); + NIR_PASS(progress, nir, nir_opt_algebraic); + NIR_PASS(progress, nir, nir_opt_dce); + /* A pass that keeps reporting progress without removing any + * control flow would spin here, and a display server starting + * up is the worst place to discover that. */ + } while (progress && ++round < 16); +} + +/* pipe_context::create_fs_state and ::create_vs_state. + * + * Gallium hands over a token stream, so this is where the decoder and the + * compiler meet Mesa. A shader that does not compile returns NULL rather than + * a broken object: by bind time the caller no longer knows which shader it + * was, and by draw time neither does anything else. */ +static int sgx_fs_fits(const struct sgx_shader *sh, char *why, size_t n); +static void *sgx_pipe_create_shader(struct pipe_context *pc, + const struct pipe_shader_state *st, + int stage); +static int sgx_vs_add_epilogue(struct tgsi_insn *insns, unsigned *n, + unsigned cap, unsigned ntemp, int mte); + +/* Why the shader was refused, for the caller that asked to be told. Mesa + * turns this into a link failure, which is the answer glamor is written for: + * it retries its composite shader without the repeat sampler and falls back + * to software for what it cannot get through. Returning NULL without a reason + * instead left every draw of a refused program silently blank. */ +static void sgx_shader_refuse(const struct pipe_shader_state *st, + const char *why) +{ + /* Said whatever the caller asked for. The hardware vertex compile does + * not ask - it takes the refusal as "the draw module runs this one" - + * so every reason a vertex program was kept off the part used to go + * nowhere, and "no hardware form" was all that could be seen. */ + sgx_dbg("shader refused: %s\n", why); + if (!st || !st->report_compile_error || st->error_message) + return; + ((struct pipe_shader_state *)st)->error_message = strdup(why); +} + + +/* Which input a sample takes its coordinate from, and which two of that + * input's components carry it. Follows the temporary nir_to_tgsi routes the + * coordinate through - the rule tex_coord_is_input() applies - but keeps the + * components rather than insisting they are already x and y. */ +/* IF, ELSE, ENDIF, BRK, CONT, BGNLOOP, ENDLOOP in Mesa's numbering. */ +static int sgx_fs_is_cf(unsigned op) +{ + return op == 74u || op == 77u || op == 78u || op == 73u || + op == 96u || op == 99u || op == 101u; +} + +/* skip_unread says the sample has channels of the temporary it does not + * read, so a write to those is none of the coordinate's business and the + * search may step over it. A TEX reads its target's components and nothing + * else; a TXB or TXL reads the level or bias from w, which sgx_fs_lod_src() + * takes separately. A TXP is the exception - w is its projector, so it has + * no spare channel at all, and folding one onto its varying without the + * projector left every pixel reading a single texel and the projective + * cases came back a flat corner colour. + * + * Getting this too narrow costs as much as too wide: a plain sample whose + * temporary has a third channel written stays unfolded, and an unfolded + * coordinate is packed to f16 in a temporary - the form measured to leave + * the sampler at texel (0,0), which is a flat fill that looks like a + * render. That is what a shadow map came back as. */ +static int sgx_fs_coord_src(const struct tgsi_insn *insns, unsigned at, + unsigned char *c0, unsigned char *c1, + int skip_unread) +{ + const struct tgsi_src *c = &insns[at].src[0]; + unsigned char s0, s1, s2; + unsigned i; + int in = -1; + unsigned writes = 0; + /* A volume or cube sample reads a third component, which the move + * has to define too and the fold has to carry. */ + int three = tgsi_tex_coord_dim(insns[at].tex_target) == 3; + + if (c->negate || c->absolute) + return -1; + s0 = (unsigned char)(c->swizzle & 3u); + s1 = (unsigned char)((c->swizzle >> 2) & 3u); + s2 = (unsigned char)((c->swizzle >> 4) & 3u); + if (c->file == TGSI_F_INPUT) { + *c0 = s0; + *c1 = s1; + return (int)c->index; + } + if (c->file != TGSI_F_TEMPORARY) + return -1; + /* The write that reaches this read, not every write that ever named + * the same index. Mesa's TGSI recycles temporary numbers - a + * lightmapped surface builds both of its coordinates in the same one, + * because the first is dead by the time the second is built - so + * scanning the whole prefix found two different varyings and gave up. + * The sample then stayed unfolded, its dead move kept reading the + * varying, and the program was put on the path that costs this part + * 450 ns a shaded pixel against 25. That is ioquake3's whole world: + * 3996 draws on the slow path against 120 on the fast one. */ + i = at; + while (i--) { + const struct tgsi_insn *w = &insns[i]; + + /* Which write reaches a read is not a question this can answer + * across a branch. */ + if (sgx_fs_is_cf(w->opcode)) + return -1; + if (w->dst.file != TGSI_F_TEMPORARY || + w->dst.index != c->index) + continue; + /* A write to the other channels - the level or bias a TXB + * or TXL carries in w - is not the coordinate's. */ + if (skip_unread && + !(w->dst.writemask & ((1u << s0) | (1u << s1) | + (three ? 1u << s2 : 0u)))) + continue; + /* It has to define both components the sample reads. */ + if ((w->dst.writemask & ((1u << s0) | (1u << s1))) != + ((1u << s0) | (1u << s1))) + return -1; + if (three && !(w->dst.writemask & (1u << s2))) + return -1; + if (w->opcode != 1 || w->nsrc != 1 || + w->src[0].file != TGSI_F_INPUT || + w->src[0].negate || w->src[0].absolute) + return -1; + /* The move's swizzle read through the sample's: the sample + * names channels of the temporary, the move says which input + * component each of those holds. */ + *c0 = (unsigned char)((w->src[0].swizzle >> (2u * s0)) & 3u); + *c1 = (unsigned char)((w->src[0].swizzle >> (2u * s1)) & 3u); + /* The third has to follow the second in the varying: the set + * carries the coordinate as three consecutive floats and the + * record moves only a pair to the front. */ + if (three && + ((w->src[0].swizzle >> (2u * s2)) & 3u) != *c1 + 1u) + return -1; + return (int)w->src[0].index; + } + (void)in; + (void)writes; + return -1; +} + + +/* Lay a sampled varying out as the record will carry it. + * + * A varying the unit samples and the program also reads is given one set of + * four floats: the coordinate, then the one component the program reads, then + * a divisor of one. The unit takes the coordinate off the front and the + * program is handed the same four registers, so both arrive. The program's + * reads are rewritten to the new positions here and sgx_pipe_vbuf.c fills the + * record from the same order, so the two cannot disagree. + * + * Returns 1 when it laid one out, with c[] naming the three original + * components in record order. */ +static int sgx_fs_lay_coord(struct tgsi_insn *insns, unsigned n, int *in_out, + unsigned char *c) +{ + unsigned char c0 = 0, c1 = 0, used = 0, inv[4], keep = 0; + int in = -1; + unsigned i, k, j; + + *in_out = -1; + /* The same switch the set description is built under: this rewrites + * the program's reads, and doing it while the record is written the + * other way round leaves the two disagreeing - the cube renders black + * and reports sixty frames a second doing it. */ + if (!SGX_ENVS("SGX_COORD_SPLIT")) + return 0; + for (i = 0; i < n; i++) { + unsigned char a = 0, b = 0; + int t; + + if (insns[i].opcode != 52 || !insns[i].nsrc) + continue; + t = sgx_fs_coord_src(insns, i, &a, &b, 0); + if (t < 0) + return 0; + if (in >= 0 && (in != t || a != c0 || b != c1)) + return 0; + in = t; + c0 = a; + c1 = b; + } + if (in < 0 || c0 == c1) + return 0; + /* What the program reads of it besides the coordinate. */ + for (i = 0; i < n; i++) + for (k = 0; k < insns[i].nsrc; k++) { + const struct tgsi_src *sr = &insns[i].src[k]; + + if (k == 0 && insns[i].opcode == 52) + continue; + if (sr->file != TGSI_F_INPUT || + sr->index != (unsigned)in) + continue; + for (j = 0; j < 4; j++) + used |= (unsigned char) + (1u << ((sr->swizzle >> (2u * j)) & 3u)); + } + used &= (unsigned char)~((1u << c0) | (1u << c1)); + if (!used) + return 0; /* a coordinate and nothing else */ + /* One spare slot, so one component besides the coordinate. */ + for (j = 0; j < 4; j++) + if (used & (1u << j)) { + if (used & ~(unsigned char)(1u << j)) + return 0; + keep = (unsigned char)j; + } + /* The coordinate, then what the program reads. A set this wide is + * sampled with texdim 2, so nothing is a divisor and the third float + * is free; the fourth is not delivered. */ + c[0] = c0; + c[1] = c1; + c[2] = keep; + inv[c0] = 0; + inv[c1] = 1; + inv[keep] = 2; + for (j = 0; j < 4; j++) + if (j != c0 && j != c1 && j != keep) + inv[j] = 3; + for (i = 0; i < n; i++) + for (k = 0; k < insns[i].nsrc; k++) { + struct tgsi_src *sr = &insns[i].src[k]; + unsigned char sw = 0; + + if (sr->file != TGSI_F_INPUT || + sr->index != (unsigned)in) + continue; + for (j = 0; j < 4; j++) + sw |= (unsigned char) + (inv[(sr->swizzle >> (2u * j)) & 3u] << + (2u * j)); + sr->swizzle = sw; + } + *in_out = in; + sgx_dbg("coord: varying %d laid out %u%u%u\n", in, c[0], c[1], + c[2]); + return 1; +} + +/* Name the varying the sample's coordinate comes from, rather than the + * temporary nir_to_tgsi moved it through. + * + * The iterated path does not read the coordinate at all - the texture unit + * has already sampled and the texel arrives in a primary attribute - but it + * checks that the coordinate is the set the unit iterated, and a temporary + * is not something it can check. Every sample whose coordinate is a plain + * move of a varying's first two components is therefore that varying, and + * saying so is what lets the iterator serve it. */ +/* The moves the fold above left behind. Their temporary was the sample's + * coordinate and nothing reads it any more, but frag_input_used() still counts + * their read of the varying - which says the program reads the coordinate for + * something other than sampling, and puts it on the shader-issued path. That + * path carries no texture issue and costs this part 450 ns a shaded pixel + * against 25, so ioquake3 drew a full-screen quad in 138 ms and its cinematic + * ran at one frame a second. */ +static unsigned sgx_fs_drop_dead_moves(struct tgsi_insn *insns, unsigned n) +{ + unsigned i, j, k, out = 0; + + if (SGX_ENVS("SGX_NO_DEAD_MOVE")) + return n; + for (i = 0; i < n; i++) { + int dead = insns[i].opcode == 1 && + insns[i].dst.file == TGSI_F_TEMPORARY; + + /* Forward to the first read or the first overwrite, not + * "is this register named anywhere". A sample writes its + * result into the very temporary its coordinate came from - + * "TXP TEMP[0], TEMP[0]" is what Mesa emits - so the whole + * program still names that register, while what the move put + * there died at the sample. Testing for any mention kept every + * one of these moves alive, and with it the read of the + * varying that forces the slow path. */ + for (j = i + 1; j < n && dead; j++) { + if (sgx_fs_is_cf(insns[j].opcode)) { + dead = 0; + break; + } + for (k = 0; k < insns[j].nsrc; k++) + if (insns[j].src[k].file == TGSI_F_TEMPORARY && + insns[j].src[k].index == + insns[i].dst.index) { + dead = 0; + break; + } + if (!dead) + break; + if (insns[j].dst.file == TGSI_F_TEMPORARY && + insns[j].dst.index == insns[i].dst.index && + (insns[j].dst.writemask & + insns[i].dst.writemask) == + insns[i].dst.writemask) + break; /* overwritten before it was read */ + } + if (!dead) + insns[out++] = insns[i]; + else + sgx_dbg("coord: the move to temp %u is dead now\n", + insns[i].dst.index); + } + return out; +} + +/* The value a TXB or TXL carries in its coordinate's w: the move that wrote + * that channel of the temporary, so the level or bias can be read from its + * source once the coordinate itself has been folded onto the varying. Fails + * across a branch, or on anything but a plain move of a uniform, an + * immediate, an input or another temporary. */ +static int sgx_fs_lod_src(const struct tgsi_insn *insns, unsigned at, + struct tgsi_src *out) +{ + const struct tgsi_src *c = &insns[at].src[0]; + unsigned char sw = (unsigned char)((c->swizzle >> 6) & 3u), ch; + unsigned i = at; + + if (c->file != TGSI_F_TEMPORARY) + return -1; + while (i--) { + const struct tgsi_insn *w = &insns[i]; + + if (sgx_fs_is_cf(w->opcode)) + return -1; + if (w->dst.file != TGSI_F_TEMPORARY || + w->dst.index != c->index || + !(w->dst.writemask & (1u << sw))) + continue; + if (w->opcode != 1 || w->nsrc != 1 || + (w->src[0].file != TGSI_F_CONSTANT && + w->src[0].file != TGSI_F_IMMEDIATE && + w->src[0].file != TGSI_F_INPUT && + w->src[0].file != TGSI_F_TEMPORARY)) + return -1; + *out = w->src[0]; + ch = (unsigned char)((w->src[0].swizzle >> (2u * sw)) & 3u); + out->swizzle = (unsigned char)(ch | (ch << 2) | (ch << 4) | + (ch << 6)); + return 0; + } + return -1; +} + +static void sgx_fs_fold_coord(struct tgsi_insn *insns, unsigned n) +{ + unsigned i; + + if (SGX_ENVS("SGX_NO_COORD_FOLD")) + return; + for (i = 0; i < n; i++) { + unsigned char c0 = 0, c1 = 0; + struct tgsi_src lod; + int in, lodop = insns[i].opcode == 68 || insns[i].opcode == 72; + /* Every form but the projective one has channels the sample + * does not read. */ + int spare = insns[i].opcode != 54; + + /* A projected sample too. Its coordinate reaches the sample + * the same way, and naming the varying is what lets the + * iterator deliver it - which is the only way this part + * samples at all, and the only way a projected one can be + * done, because the divide happens on the way in. A biased + * sample and one at a level (TXB, TXL) likewise: the SMP + * takes the coordinate from the iterator and the level from + * a register, so the two halves of the temporary Mesa built + * go their separate ways - measured, the packed-temporary + * form read the wrong level for every bias. */ + if ((insns[i].opcode != 52 && insns[i].opcode != 54 && + !lodop) || !insns[i].nsrc || + insns[i].src[0].file != TGSI_F_TEMPORARY) + continue; + if (lodop && sgx_fs_lod_src(insns, i, &lod)) + continue; + in = sgx_fs_coord_src(insns, i, &c0, &c1, spare); + if (in < 0) + continue; + if (lodop) { + insns[i].src[2] = lod; + insns[i].nsrc = 3; + } + /* Only a coordinate that leads its varying. One Mesa packed + * behind a scalar has been laid out to lead by + * sgx_fs_lay_coord() already; one that has not cannot be + * served by the iterator, because the set it would name is + * three floats projected and the coordinate would be divided + * by whatever shares it. Those keep sampling for themselves, + * which is slower and right. */ + /* Any pair, not only the leading one: the record moves a + * set's coordinate to its front, so a coordinate Mesa packed + * at .zw is served by a set of its own. */ + if (c0 == c1) + continue; + memset(&insns[i].src[0], 0, sizeof insns[i].src[0]); + insns[i].src[0].file = TGSI_F_INPUT; + insns[i].src[0].index = (unsigned)in; + /* The components the coordinate is in, kept: the shader-issued + * path samples with them, and the iterated path reads them off + * this operand to know which two of the varying the unit's own + * set has to carry - three for a volume, the third one on + * from the second, which sgx_fs_coord_src() required. */ + if (tgsi_tex_coord_dim(insns[i].tex_target) == 3) + insns[i].src[0].swizzle = (unsigned char) + (c0 | (c1 << 2) | ((c1 + 1u) << 4) | + ((c1 + 1u) << 6)); + else + insns[i].src[0].swizzle = (unsigned char) + (c0 | (c1 << 2) | (c1 << 4) | (c1 << 6)); + sgx_dbg("coord: sample %u reads varying %d.%u%u directly\n", i, + in, c0, c1); + } +} + +/* Which sampler units the program reads through a shadow sampler. */ +static unsigned sgx_nir_shadow_samplers(nir_shader *s) +{ + unsigned m = 0; + + nir_foreach_function_impl(impl, s) { + nir_foreach_block(block, impl) { + nir_foreach_instr(instr, block) { + nir_tex_instr *tex; + + if (instr->type != nir_instr_type_tex) + continue; + tex = nir_instr_as_tex(instr); + if (tex->is_shadow) + m |= 1u << (tex->sampler_index & 31u); + } + } + } + return m; +} + +/* Lower every shadow sample to a plain one and the comparison the bound + * sampler asks for, read through the view's swizzle - what GL's + * sampler2DShadow means on a part whose unit compares nothing. A unit + * with no comparison bound compares ALWAYS, which is what GL leaves + * undefined. The reference is clamped to [0, 1] first, as it is for a + * fixed-point depth texture, which is what the caller declared. */ +static void sgx_nir_lower_shadow(struct sgx_pipe_context *c, nir_shader *s) +{ + unsigned n = c->ncmp ? c->ncmp : 1u, i; + enum compare_func *func = malloc(n * sizeof *func); + nir_lower_tex_shadow_swizzle *swz = malloc(n * sizeof *swz); + + if (!func || !swz) { + free(func); + free(swz); + return; + } + for (i = 0; i < n; i++) { + uint32_t k = i < c->ncmp ? c->cmp[i] : 0u; + uint16_t sw = k ? SGX_CMPKEY_SWZ(k) : SGX_SWZ_IDENTITY; + + func[i] = k ? (enum compare_func)SGX_CMPKEY_FUNC(k) : + COMPARE_FUNC_ALWAYS; + swz[i].swizzle_r = SGX_SWZ_CHAN(sw, 0); + swz[i].swizzle_g = SGX_SWZ_CHAN(sw, 1); + swz[i].swizzle_b = SGX_SWZ_CHAN(sw, 2); + swz[i].swizzle_a = SGX_SWZ_CHAN(sw, 3); + } + NIR_PASS(_, s, nir_lower_tex_shadow, n, func, swz, true); + free(func); + free(swz); +} + +/* Build a fragment program again from its NIR for the comparison now bound + * on its shadow samplers, in place: the object's address is what the + * context and the caller hold. */ +static int sgx_pipe_fs_rebuild_cmp(struct sgx_pipe_context *c, + struct sgx_shader *sh) +{ + struct pipe_shader_state conv; + struct sgx_shader *fresh; + nir_shader *nir; + void *keep_nir = sh->src_nir; + unsigned keep_shadow = sh->shadow_mask; + uint32_t *key; + unsigned n = c->ncmp; + + if (!keep_nir) + return -1; + nir = nir_shader_clone(NULL, keep_nir); + if (!nir) + return -1; + key = n ? malloc(n * sizeof *key) : NULL; + if (n && !key) { + ralloc_free(nir); + return -1; + } + if (n) + memcpy(key, c->cmp, n * sizeof *key); + sgx_nir_lower_shadow(c, nir); + memset(&conv, 0, sizeof conv); + conv.type = PIPE_SHADER_IR_TGSI; + conv.tokens = nir_to_tgsi(nir, c->base.screen); + if (!conv.tokens) { + free(key); + return -1; + } + /* Compiled as a fresh object, and its contents moved over. */ + sh->src_nir = NULL; + fresh = sgx_pipe_create_shader(&c->base, &conv, UIR_STAGE_FRAGMENT); + FREE((void *)conv.tokens); + free(conv.error_message); + if (!fresh) { + sh->src_nir = keep_nir; + free(key); + return -1; + } + sgx_shader_fini(sh); + free(sh->src_insns); + free(sh->src_io); + free(sh->swz_key); + free(sh->cls_key); + free(sh->cmp_key); + *sh = *fresh; + free(fresh); + sh->src_nir = keep_nir; + sh->shadow_mask = keep_shadow; + sh->cmp_key = key; + sh->ncmp_key = n; + return 0; +} + +static void *sgx_pipe_create_shader(struct pipe_context *pc, + const struct pipe_shader_state *st, + int stage) +{ + /* The compile block below names its status `st` as well. */ + const struct pipe_shader_state *shst = st; + struct sgx_shader *sh; + /* Grown to fit the program rather than fixed at 256: a shader with + * more instructions than that was refused outright, and lighting or + * bump-mapping shaders go well past it. The epilogue needs room on top + * of what was parsed, which is what the slack below is for. */ + struct tgsi_insn *insns = NULL; + unsigned insn_cap = 0; + struct tgsi_io io; + unsigned n = 0; + int tstage = 0; + + sgx_dbg("create shader: stage %d, ir %d\n", stage, st ? (int)st->type : -1); + if (!st) + return NULL; + /* mesa/st hands drivers NIR now, so the tokens this compiler takes + * come from nir_to_tgsi rather than from the state tracker. The tokens + * are freed below: they exist only to be compiled. */ + if (st->type == PIPE_SHADER_IR_NIR && st->ir.nir) { + struct pipe_shader_state conv = *st; + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + nir_shader *keep = NULL; + unsigned shadow = 0; + void *sh; + + /* The stipple stage's variant of a program samples its + * pattern on the unit after the program's own, and the frame + * issues SGX_MAX_TEX_UNITS; a unit past that reports itself + * and samples black, which would stipple everything away. The + * stage draws that program unstippled when this refuses. */ + if (stage == UIR_STAGE_FRAGMENT) { + nir_foreach_uniform_variable(var, st->ir.nir) { + if (!glsl_type_is_sampler(var->type) || + !var->name || + strcmp(var->name, "stipple_tex")) + continue; + if (var->data.binding < SGX_MAX_TEX_UNITS) { + ((struct sgx_pipe_context *)pc)-> + stipple_unit = + (int)var->data.binding; + continue; + } + { + static int said; + + if (!said++) + fprintf(sgx_log(), "sgx: " + "glPolygonStipple on a " + "program using every " + "texture unit: it wants " + "unit %u of %u and " + "draws unstippled\n", + var->data.binding, + (unsigned)SGX_MAX_TEX_UNITS); + } + ralloc_free(st->ir.nir); + return NULL; + } + } + conv.type = PIPE_SHADER_IR_TGSI; + /* A shadow sampler's comparison is not in the unit. It is + * lowered here, for the comparison bound now, and the NIR is + * kept so another can be lowered later - ARB_shadow, which + * Mesa advertises whatever the driver says, and the ES2 + * depth texture with it. */ + if (stage == UIR_STAGE_FRAGMENT && + (shadow = sgx_nir_shadow_samplers(st->ir.nir)) != 0) { + keep = nir_shader_clone(NULL, st->ir.nir); + sgx_nir_lower_shadow(c, st->ir.nir); + } + /* A 1D sample becomes a 2D one at t = 0.5 - row zero of the + * one-row surface under every wrap mode - because the unit + * has no 1D type and the frontend reads every TEX as 2D. */ + { + nir_lower_tex_options lo = { .lower_1d = true }; + + NIR_PASS(_, st->ir.nir, nir_lower_tex, &lo); + } + sgx_nir_flatten_cf(st->ir.nir); + conv.tokens = nir_to_tgsi(st->ir.nir, pc->screen); + if (!conv.tokens) { + sgx_dbg("stage %d: nir_to_tgsi produced nothing\n", stage); + ralloc_free(keep); + return NULL; + } + if (sgx_debug()) { + const uint32_t *t = (const uint32_t *)conv.tokens; + unsigned k, ntok = (t[0] & 0xffu) + ((t[0] >> 8) & 0xffffffu); + + fprintf(sgx_log(), "sgx: nir_to_tgsi produced %u tokens:\n", ntok); + for (k = 0; k < ntok && k < 64; k++) + fprintf(sgx_log(), " [%2u] %08x\n", k, t[k]); + tgsi_dump(conv.tokens, 0); + } + sh = sgx_pipe_create_shader(pc, &conv, stage); + FREE((void *)conv.tokens); + /* The refusal was recorded against the converted tokens; the + * caller waiting for it holds the original state object. */ + if (!sh && conv.error_message) + ((struct pipe_shader_state *)st)->error_message = + conv.error_message; + if (sh && keep) { + struct sgx_shader *fs = sh; + + fs->src_nir = keep; + fs->shadow_mask = shadow; + fs->cmp_key = c->ncmp ? + malloc(c->ncmp * sizeof *fs->cmp_key) : + NULL; + if (fs->cmp_key) { + memcpy(fs->cmp_key, c->cmp, + c->ncmp * sizeof *fs->cmp_key); + fs->ncmp_key = c->ncmp; + } + } else { + ralloc_free(keep); + } + return sh; + } + if (st->type != PIPE_SHADER_IR_TGSI || !st->tokens) { + sgx_dbg("stage %d: shader is neither NIR nor TGSI\n", stage); + return NULL; + } + + /* The length comes from the header, not from a guess. tgsi_header + * carries HeaderSize in bits [7:0] and BodySize in [31:8], and the + * total is header plus body. Passing a fixed upper bound instead reads + * past the end of whatever Mesa allocated - which is what the first + * version of this did. */ + { + const uint32_t *tok = (const uint32_t *)st->tokens; + unsigned hdr = tok[0] & 0xffu; + unsigned body = (tok[0] >> 8) & 0xffffffu; + unsigned ntok = hdr + body; + + if (!hdr || ntok < 2 || ntok > 65536) { + sgx_dbg("stage %d: a %u token header is not one\n", + stage, ntok); + return NULL; + } + /* One token cannot become more than one instruction, so the + * token count bounds what the parse can produce; the slack is + * the viewport epilogue's. */ + insn_cap = ntok + 8 + TGSI_PARSE_MAX_IO; + insns = calloc(insn_cap, sizeof *insns); + if (!insns) + return NULL; + { + int pr = tgsi_parse_tokens_io(tok, ntok, insns, + insn_cap, &n, &tstage, + &io); + + if (pr != TGSI_PARSE_OK) { + /* Unconditionally: refusing the shader means + * every draw that uses it is skipped and the + * picture is silently missing whatever it drew. + * Behind SGX_DEBUG this looked like a driver + * that renders nothing for no reason. */ + fprintf(sgx_log(), "sgx: stage %d: %u tokens did " + "not parse: %s - the shader is refused " + "and its draws will not appear\n", + stage, ntok, tgsi_parse_status_name(pr)); + free(insns); + return NULL; + } + } + } + + /* A vertex program has to end in window coordinates; the shader ends + * in clip space. */ + if (stage == UIR_STAGE_VERTEX) { + unsigned ntemp = 0, k, s; + + for (k = 0; k < n; k++) { + if (insns[k].dst.file == TGSI_F_TEMPORARY && + insns[k].dst.index + 1 > ntemp) + ntemp = insns[k].dst.index + 1; + for (s = 0; s < insns[k].nsrc; s++) + if (insns[k].src[s].file == TGSI_F_TEMPORARY && + insns[k].src[s].index + 1 > ntemp) + ntemp = insns[k].src[s].index + 1; + } + /* With the MTE doing the perspective divide and the viewport + * map, the program emits clip space and needs no epilogue - + * which is also what frees the record dwords the viewport was + * passed in. That is what the vendor's captured hardware + * program does: o0..o3 = MVP * pos, then emitvtx, with no + * reciprocal anywhere in it. */ + if (!sgx_vs_add_epilogue(insns, &n, insn_cap, ntemp, + sgx_mte_viewport())) { + sgx_dbg("stage 0: no room for the viewport epilogue\n"); + free(insns); + return NULL; + } + } + + sh = calloc(1, sizeof(*sh)); + if (!sh) { + free(insns); + return NULL; + } + unsigned char coord_c[3] = { 0, 1, 2 }; + int coord_in = -1; + + if (stage == UIR_STAGE_FRAGMENT) { + (void)sgx_fs_lay_coord(insns, n, &coord_in, coord_c); + sgx_fs_fold_coord(insns, n); + n = sgx_fs_drop_dead_moves(insns, n); + } + /* What the backend is handed, which is what its refusals name. The + * TGSI Mesa prints is not this: the driver parses it into its own + * form, and a coordinate the iterator could serve or not is decided on + * these operands. */ + if (SGX_ENVS("SGX_DUMP_TGSI")) { + unsigned k, s2; + + fprintf(sgx_log(), "sgx: stage %d, %u instruction(s):\n", stage, n); + for (k = 0; k < n; k++) { + fprintf(sgx_log(), "sgx: %3u: op %3u dst f%u[%u].%x", + k, insns[k].opcode, insns[k].dst.file, + insns[k].dst.index, insns[k].dst.writemask); + for (s2 = 0; s2 < insns[k].nsrc; s2++) + fprintf(sgx_log(), " src%u f%u[%u] swz %02x%s%s", + s2, insns[k].src[s2].file, + insns[k].src[s2].index, + insns[k].src[s2].swizzle, + insns[k].src[s2].negate ? " neg" : "", + insns[k].src[s2].absolute ? " abs" : ""); + fprintf(sgx_log(), "\n"); + } + } + { + char cerr[256] = ""; + enum sgx_shader_status st; + + /* The iterator serves two texture units and the shader-issued + * path does not, so try the iterator first and only fall back + * when the program cannot be expressed that way - a projected + * or computed coordinate is not something it can apply. */ + /* SGX_NO_ITER_FIRST goes back to letting each program's own + * shape choose the path, which accelerates more of them but + * draws the X server's text as solid boxes. */ + int iter_first = SGX_ENVS("SGX_NO_ITER_FIRST") == NULL; + struct sgx_pipe_context *fc = (struct sgx_pipe_context *)pc; + + sgx_shader_compile_lock(); + sgx_shader_face_swap(fc->cur_rast ? (int)sgx_face_swap_of( + &((const struct sgx_pipe_rast *)fc->cur_rast)->hw) : 0); + sgx_shader_prefer_preiterated(iter_first ? 1 : -1); + st = sgx_shader_from_tgsi_io(sh, stage, insns, n, NULL, 0, + &io, cerr, sizeof cerr); + /* A program the iterator cannot serve goes to the + * shader-issued path. Refusing it outright instead was meant + * to send the caller elsewhere, but there is nowhere to go: + * the bind leaves no fragment program, sgx_draw() refuses + * every draw that follows and the frame renders nothing - + * which is why ioquake3, whose fixed-function shaders sample + * with TXP, drew a black window while reporting no faults. + * + * What that refusal was protecting against is a masked + * composite sampled on this path, which delivers the mask as + * opaque and leaves a solid box where a glyph should be. The + * ntex > 1 check below still refuses exactly those, and glamor + * retries without the repeat sampler, so the desktop keeps + * what it had. SGX_NO_SMP_FALLBACK restores the old refusal. */ + if (st != SGX_SHADER_OK && stage == UIR_STAGE_FRAGMENT && + !SGX_ENVS("SGX_NO_SMP_FALLBACK")) { + struct sgx_shader *sh2 = calloc(1, sizeof(*sh2)); + + /* The reason the iterator could not serve it, which + * was thrown away here: the retry overwrites cerr, so + * a program that ends up sampling for itself gave no + * account of why. The two paths are not close in cost + * - a shader-issued sample put glmark2's textured cube + * at 1.4 us a pixel - so which programs land on this + * one, and for what reason, is worth saying. */ + sgx_dbg("the iterator cannot serve this program, so it " + "samples for itself: %s\n", cerr); + + sgx_shader_fini(sh); + free(sh); + sh = sh2; + if (!sh) { + free(insns); + sgx_shader_prefer_preiterated(-1); + sgx_shader_compile_unlock(); + return NULL; + } + sgx_shader_prefer_preiterated(0); + st = sgx_shader_from_tgsi_io(sh, stage, insns, n, NULL, + 0, &io, cerr, sizeof cerr); + } + sgx_shader_prefer_preiterated(-1); + sgx_shader_face_swap(0); + sgx_shader_compile_unlock(); + + if (st != SGX_SHADER_OK) { + char why[320]; + + snprintf(why, sizeof why, + "sgx: %u instructions did not compile: %s%s%s", + n, sgx_shader_status_name(st), + cerr[0] ? " - " : "", cerr); + /* A program that does not compile is a draw that will + * not render, so it is said unconditionally rather + * than only under SGX_DEBUG - the symptom is a blank + * window, and the reason belongs in the log. */ + fprintf(sgx_log(), "sgx: stage %d: %u TGSI instructions " + "did not compile: %s%s%s\n", stage, n, + sgx_shader_status_name(st), + cerr[0] ? " - " : "", cerr); + sgx_shader_refuse(shst, why); + sgx_shader_fini(sh); + free(sh); + free(insns); + return NULL; + } + } + if (sh->coord_split_in >= 0) { + unsigned q; + + for (q = 0; q < 3; q++) + sh->coord_c[q] = coord_c[q]; + } + /* Two units on this path were refused because the second was thought + * to come back as whatever the first held. It does not: a program + * with two textures, a red one and a green one added, and a + * coordinate computed so the iterator cannot deliver it, renders both + * correctly twenty times of twenty. What was really refusing those + * composites was the register budget - the X server's glyph shader + * wants thirty-two registers for its values where a task can name + * thirty-one - and a refusal here only hid it. + * + * So they are let through. SGX_ITER_FIRST=1 puts the refusal back. */ + if (stage == UIR_STAGE_FRAGMENT && sh->compiled && + !sh->tex_preiterated && SGX_ENVS("SGX_ITER_FIRST")) { + unsigned u, ntex = 0; + + /* Indexed by varying, counting the samplers that read it. */ + for (u = 0; u < sizeof sh->tex_units / sizeof sh->tex_units[0]; + u++) + ntex += sh->tex_units[u]; + if (ntex > 1) { + sgx_dbg("stage %d: %u units sampled by the program " + "itself\n", stage, ntex); + sgx_shader_refuse(shst, + "sgx: the program samples more than " + "one texture for itself"); + sgx_shader_fini(sh); + free(sh); + free(insns); + return NULL; + } + } + /* A fragment program needing more temporaries than the task's count + * can name never runs: the frame is refused at submit and the draw + * disappears silently. Refusing it here fails the link instead, which + * is the answer callers are written for - glamor retries its composite + * shader without the repeat sampler when the link fails. */ + if (stage == UIR_STAGE_FRAGMENT) { + char why[192]; + + if (!sgx_fs_fits(sh, why, sizeof why)) { + sgx_dbg("stage %d: %s\n", stage, why); + sgx_shader_refuse(shst, why); + sgx_shader_fini(sh); + free(sh); + free(insns); + return NULL; + } + } + /* The secondary attribute bank holds thirty-two registers. Past that + * the program renders differently run to run and nothing is logged: + * measured with a shader whose only variable is how many constants it + * reads, ten runs each - sa 28 and sa 32 pass ten of ten, sa 36 passes + * one, sa 44 passes one. The X server's masked composite asks for + * thirty-six, which is exactly why it has always been flaky. + * + * Refused here rather than left to corrupt: a failed link sends the + * caller somewhere that works, where a frame that renders differently + * every time cannot be worked around at all. */ + /* A fragment program is kept in the form it arrived in: an sRGB view + * bound later changes what the sample has to compute, and the program + * is then built again from this rather than refused. */ + if (stage == UIR_STAGE_FRAGMENT) { + struct tgsi_io *sio = malloc(sizeof *sio); + + if (sio) { + *sio = io; + sh->src_insns = insns; + sh->src_n = n; + sh->src_io = sio; + } else { + free(insns); + } + } else { + free(insns); + } + sgx_dbg("stage %d: %u TGSI in, %u USSE out, %u temps\n", stage, n, + sh->ninsns, sh->ntemps); + return sh; +} + +/* Rebuild a fragment program for the sampler units that now carry an sRGB + * view. Nothing to do for the usual case of none, which is why the key is + * checked before anything is touched. Returns 0 when the program in hand is + * the right one. */ +/* What a fragment program has to fit for the hardware to be told about it. + * Non-zero when it does; otherwise why says which budget it passed. + * + * Applied at create, and again after an sRGB retarget: that rebuild adds + * instructions and registers to a program that was already accepted, so one + * sitting just under either budget passed create and then overflowed with + * nothing checking. */ +static int sgx_fs_fits(const struct sgx_shader *sh, char *why, size_t n) +{ + if (!sgx_temp_ok(sh->ntemps)) { + snprintf(why, n, "sgx: the program needs %u temporaries and " + "a task can name %u", sh->ntemps, sgx_temp_max()); + return 0; + } + if (sh->nsecattr > sgx_sa_max()) { + snprintf(why, n, "sgx: the program needs %u secondary " + "attributes and the bank holds %u (%u uniform, " + "samplers at %u, literals at %u)", sh->nsecattr, + sgx_sa_max(), sh->nuniform, sh->smp_base, + sh->pool_base); + return 0; + } + return 1; +} + +/* Rebuild a fragment program for the swizzles the bound views ask for. The + * same shape as the sRGB rebuild, and for the same reason: the part has no + * texture swizzle, so what GL_ALPHA and GL_LUMINANCE mean is baked into the + * program and changes with the view. */ +/* The swizzles a program is compiled with: the views' own, except on a + * shadow sampler, whose swizzle its comparison's lowering already applied. + * buf holds n entries. */ +static const uint16_t *swz_less_cmp(const struct sgx_shader *sh, + const uint16_t *swz, unsigned n, + uint16_t *buf) +{ + unsigned i; + + if (!sh->ncmp_key || !swz) + return swz; + for (i = 0; i < n; i++) + buf[i] = i < sh->ncmp_key && sh->cmp_key[i] ? + SGX_SWZ_IDENTITY : swz[i]; + return buf; +} + +int sgx_shader_retarget_swz(struct sgx_shader *sh, const uint16_t *swz, + unsigned n) +{ + char cerr[256] = ""; + enum sgx_shader_status st; + uint16_t *buf; + + if (!sh || sh->stage != UIR_STAGE_FRAGMENT) + return 0; + if (!sgx_shader_swz_differs(sh, swz, n)) + return 0; + if (!sh->src_insns || !sh->src_io) + return -1; + buf = n ? malloc(n * sizeof *buf) : NULL; + if (n && !buf) + return -1; + sgx_shader_compile_lock(); + sgx_shader_swz_units(swz_less_cmp(sh, swz, n, buf), n); + sgx_shader_srgb_units(sh->srgb_key); + sgx_shader_depth_units(sh->depth_key); + sgx_shader_npot_units(sh->npot_key); + sgx_shader_face_swap((int)sh->face_key); + sgx_shader_cls_units(sh->cls_key, sh->ncls_key); + sgx_shader_prefer_preiterated(sh->tex_preiterated ? 1 : 0); + st = sgx_shader_from_tgsi_io(sh, UIR_STAGE_FRAGMENT, sh->src_insns, + sh->src_n, NULL, 0, sh->src_io, + cerr, sizeof cerr); + sgx_shader_prefer_preiterated(-1); + sgx_shader_face_swap(0); + sgx_shader_srgb_units(0); + sgx_shader_depth_units(0); + sgx_shader_npot_units(0); + sgx_shader_swz_units(NULL, 0); + sgx_shader_cls_units(NULL, 0); + sgx_shader_compile_unlock(); + free(buf); + if (st != SGX_SHADER_OK) { + fprintf(sgx_log(), "sgx: rebuilding for texture swizzles " + "failed: %s%s%s\n", sgx_shader_status_name(st), + cerr[0] ? " - " : "", cerr); + return -1; + } + if (!sgx_fs_fits(sh, cerr, sizeof cerr)) { + fprintf(sgx_log(), "sgx: rebuilding for texture swizzles no " + "longer fits: %s - the draw is skipped\n", cerr); + return -1; + } + return 0; +} + +/* A texture the unit cannot wrap by itself: its size is not a power of two, + * and the sampler asks for a wrap that repeats. Clamping modes are exact at + * any size and are left to the unit. */ +static int npot_axis(uint32_t size, enum sgx_wrap w, int mirror) +{ + if (!size || !(size & (size - 1u))) + return 0; + return mirror ? w == SGX_WRAP_MIRROR_REPEAT : w == SGX_WRAP_REPEAT; +} + +static unsigned sgx_npot_wrap_mask(const struct sgx_pipe_context *c) +{ + unsigned rs = 0, rt = 0, ms = 0, mt = 0, u; + + /* SGX_NO_NPOT_WRAP gives the unit's own REPEAT back, which wraps at + * the padded power of two. It is wrong for a texture that is not one, + * and it is the only way to tell this pass apart from a defect that + * merely arrived with it - which is what it was wanted for. */ + if (SGX_ENVS("SGX_NO_NPOT_WRAP")) + return 0; + for (u = 0; u < SGX_MAX_TEX_UNITS; u++) { + const struct sgx_sampler_state *st = c->cur_samplers[u]; + const struct pipe_sampler_view *pv = c->cur_views[u]; + const struct sgx_pipe_resource *r; + + if (!st || !pv || !pv->texture) + continue; + r = (const struct sgx_pipe_resource *)pv->texture; + rs |= (unsigned)npot_axis(r->r.width, st->wrap_s, 0) << u; + rt |= (unsigned)npot_axis(r->r.height, st->wrap_t, 0) << u; + ms |= (unsigned)npot_axis(r->r.width, st->wrap_s, 1) << u; + mt |= (unsigned)npot_axis(r->r.height, st->wrap_t, 1) << u; + } + return SGX_NPOT_KEY(rs, rt, ms, mt); +} + +int sgx_shader_retarget_srgb(struct sgx_shader *sh, unsigned mask) +{ + char cerr[256] = ""; + enum sgx_shader_status st; + + if (!sh || sh->stage != UIR_STAGE_FRAGMENT || sh->srgb_key == mask) + return 0; + if (!sh->src_insns || !sh->src_io) + return -1; + sgx_shader_compile_lock(); + sgx_shader_srgb_units(mask); + sgx_shader_depth_units(sh->depth_key); + sgx_shader_npot_units(sh->npot_key); + /* Rebuilding for one piece of baked-in state must not drop the other: + * without this the swizzle came back out of the program and the two + * rebuilds took turns undoing each other. */ + sgx_shader_swz_units(sh->swz_key, sh->nswz_key); + sgx_shader_face_swap((int)sh->face_key); + sgx_shader_cls_units(sh->cls_key, sh->ncls_key); + sgx_shader_prefer_preiterated(sh->tex_preiterated ? 1 : 0); + st = sgx_shader_from_tgsi_io(sh, UIR_STAGE_FRAGMENT, sh->src_insns, + sh->src_n, NULL, 0, sh->src_io, + cerr, sizeof cerr); + sgx_shader_prefer_preiterated(-1); + sgx_shader_face_swap(0); + sgx_shader_srgb_units(0); + sgx_shader_depth_units(0); + sgx_shader_npot_units(0); + sgx_shader_swz_units(NULL, 0); + sgx_shader_cls_units(NULL, 0); + sgx_shader_compile_unlock(); + if (st != SGX_SHADER_OK) { + fprintf(sgx_log(), "sgx: rebuilding for sRGB units %x failed: " + "%s%s%s\n", mask, sgx_shader_status_name(st), + cerr[0] ? " - " : "", cerr); + return -1; + } + if (!sgx_fs_fits(sh, cerr, sizeof cerr)) { + fprintf(sgx_log(), "sgx: rebuilding for sRGB units %x no longer " + "fits: %s - the draw is skipped\n", mask, cerr); + return -1; + } + return 0; +} + +/* And the same for the units carrying a depth view: the reassembly is baked + * into the program, so binding a depth texture where a colour one was rebuilds + * it. */ +int sgx_shader_retarget_depth(struct sgx_shader *sh, unsigned mask) +{ + char cerr[256] = ""; + enum sgx_shader_status st; + + if (!sh || sh->stage != UIR_STAGE_FRAGMENT || sh->depth_key == mask) + return 0; + if (!sh->src_insns || !sh->src_io) + return -1; + sgx_shader_compile_lock(); + sgx_shader_depth_units(mask); + sgx_shader_srgb_units(sh->srgb_key); + sgx_shader_npot_units(sh->npot_key); + sgx_shader_swz_units(sh->swz_key, sh->nswz_key); + sgx_shader_face_swap((int)sh->face_key); + sgx_shader_cls_units(sh->cls_key, sh->ncls_key); + sgx_shader_prefer_preiterated(sh->tex_preiterated ? 1 : 0); + st = sgx_shader_from_tgsi_io(sh, UIR_STAGE_FRAGMENT, sh->src_insns, + sh->src_n, NULL, 0, sh->src_io, + cerr, sizeof cerr); + sgx_shader_prefer_preiterated(-1); + sgx_shader_face_swap(0); + sgx_shader_srgb_units(0); + sgx_shader_depth_units(0); + sgx_shader_npot_units(0); + sgx_shader_swz_units(NULL, 0); + sgx_shader_cls_units(NULL, 0); + sgx_shader_compile_unlock(); + if (st != SGX_SHADER_OK) { + fprintf(sgx_log(), "sgx: rebuilding for depth units %x failed: " + "%s%s%s\n", mask, sgx_shader_status_name(st), + cerr[0] ? " - " : "", cerr); + return -1; + } + if (!sgx_fs_fits(sh, cerr, sizeof cerr)) { + fprintf(sgx_log(), "sgx: rebuilding for depth units %x no " + "longer fits: %s - the draw is skipped\n", mask, cerr); + return -1; + } + return 0; +} + +/* Rebuild a program for a different set of self-wrapped units. Which units + * need it depends on the bound views and samplers, not on the program, so it + * is baked in the way the sRGB mask is and rebuilt when it changes. */ +int sgx_shader_retarget_npot(struct sgx_shader *sh, unsigned mask) +{ + char cerr[256] = ""; + enum sgx_shader_status st; + + if (!sh || sh->stage != UIR_STAGE_FRAGMENT || sh->npot_key == mask) + return 0; + if (!sh->src_insns || !sh->src_io) + return -1; + sgx_shader_compile_lock(); + sgx_shader_npot_units(mask); + sgx_shader_srgb_units(sh->srgb_key); + sgx_shader_depth_units(sh->depth_key); + sgx_shader_swz_units(sh->swz_key, sh->nswz_key); + sgx_shader_face_swap((int)sh->face_key); + sgx_shader_cls_units(sh->cls_key, sh->ncls_key); + sgx_shader_prefer_preiterated(sh->tex_preiterated ? 1 : 0); + st = sgx_shader_from_tgsi_io(sh, UIR_STAGE_FRAGMENT, sh->src_insns, + sh->src_n, NULL, 0, sh->src_io, + cerr, sizeof cerr); + sgx_shader_prefer_preiterated(-1); + sgx_shader_face_swap(0); + sgx_shader_srgb_units(0); + sgx_shader_depth_units(0); + sgx_shader_npot_units(0); + sgx_shader_swz_units(NULL, 0); + sgx_shader_cls_units(NULL, 0); + sgx_shader_compile_unlock(); + if (st != SGX_SHADER_OK) { + fprintf(sgx_log(), "sgx: rebuilding for the wrap of units %x " + "failed: %s%s%s\n", mask, sgx_shader_status_name(st), + cerr[0] ? " - " : "", cerr); + return -1; + } + if (!sgx_fs_fits(sh, cerr, sizeof cerr)) { + fprintf(sgx_log(), "sgx: rebuilding for the wrap of units %x " + "no longer fits: %s - the draw is skipped\n", mask, + cerr); + return -1; + } + return 0; +} + +/* Rebuild a program reading gl_FrontFacing for the other winding: which + * device-space winding is front is baked in, the way the vendor's driver + * carries it as a uniform (opengles2/uniform.c GLSLBV_PMXSWAPFRONTFACE), and + * glFrontFace changes it. */ +int sgx_shader_retarget_face(struct sgx_shader *sh, unsigned swap) +{ + char cerr[256] = ""; + enum sgx_shader_status st; + + if (!sh || sh->stage != UIR_STAGE_FRAGMENT || sh->face_in < 0 || + sh->face_key == swap) + return 0; + if (!sh->src_insns || !sh->src_io) + return -1; + sgx_shader_compile_lock(); + sgx_shader_face_swap((int)swap); + sgx_shader_srgb_units(sh->srgb_key); + sgx_shader_depth_units(sh->depth_key); + sgx_shader_npot_units(sh->npot_key); + sgx_shader_swz_units(sh->swz_key, sh->nswz_key); + sgx_shader_cls_units(sh->cls_key, sh->ncls_key); + sgx_shader_prefer_preiterated(sh->tex_preiterated ? 1 : 0); + st = sgx_shader_from_tgsi_io(sh, UIR_STAGE_FRAGMENT, sh->src_insns, + sh->src_n, NULL, 0, sh->src_io, + cerr, sizeof cerr); + sgx_shader_prefer_preiterated(-1); + sgx_shader_face_swap(0); + sgx_shader_srgb_units(0); + sgx_shader_depth_units(0); + sgx_shader_npot_units(0); + sgx_shader_swz_units(NULL, 0); + sgx_shader_cls_units(NULL, 0); + sgx_shader_compile_unlock(); + if (st != SGX_SHADER_OK) { + fprintf(sgx_log(), "sgx: rebuilding for the front face failed: " + "%s%s%s\n", sgx_shader_status_name(st), + cerr[0] ? " - " : "", cerr); + return -1; + } + /* Over budget, and still the program every draw after this one will + * use: the key was set by the compile above, so nothing rebuilds + * again and the caller stops asking. Skipping only the draw that + * happened to trigger the rebuild is not a budget being enforced, it + * is one quad missing from the picture - which is what + * "front facing, GL_CW" was: the counter-clockwise quad, the draw + * glFrontFace changed the winding for, came back black while the + * clockwise one drawn with the very same program came back right. + * So it is reported, once, and drawn. */ + if (!sgx_fs_fits(sh, cerr, sizeof cerr)) { + static int said; + + if (!said++) + fprintf(sgx_log(), "sgx: the program rebuilt for the " + "front face is past a budget: %s - it is drawn " + "with anyway, as every later draw is\n", cerr); + } + return 0; +} + +/* Rebuild a fragment program for what the bound views' fetches return. The + * same shape again: a float or 16-bit texel is unpacked differently from + * a packed one and takes as many state blocks as it has planes, and both + * are laid down by the compiler. */ +int sgx_shader_retarget_cls(struct sgx_shader *sh, const unsigned char *cls, + unsigned n) +{ + char cerr[256] = ""; + enum sgx_shader_status st; + int preiter; + + if (!sh || sh->stage != UIR_STAGE_FRAGMENT) + return 0; + if (!sgx_shader_cls_differs(sh, cls, n)) + return 0; + if (!sh->src_insns || !sh->src_io) + return -1; + /* Read before the compile: a failed one leaves the object cleared, + * and the retry below was deciding on that cleared flag - so the + * float texture cases rebuilt once, failed, and drew nothing. */ + preiter = sh->tex_preiterated ? 1 : 0; + sgx_shader_compile_lock(); + sgx_shader_cls_units(cls, n); + sgx_shader_face_swap((int)sh->face_key); + sgx_shader_srgb_units(sh->srgb_key); + sgx_shader_depth_units(sh->depth_key); + sgx_shader_npot_units(sh->npot_key); + sgx_shader_swz_units(sh->swz_key, sh->nswz_key); + sgx_shader_prefer_preiterated(preiter); + st = sgx_shader_from_tgsi_io(sh, UIR_STAGE_FRAGMENT, sh->src_insns, + sh->src_n, NULL, 0, sh->src_io, + cerr, sizeof cerr); + /* A program the iterator served cannot deliver a float texel; it + * samples for itself from here on, as the create path would have + * had the view been bound then. */ + if (st != SGX_SHADER_OK && preiter) { + sgx_shader_prefer_preiterated(0); + st = sgx_shader_from_tgsi_io(sh, UIR_STAGE_FRAGMENT, + sh->src_insns, sh->src_n, NULL, + 0, sh->src_io, cerr, sizeof cerr); + } + sgx_shader_prefer_preiterated(-1); + sgx_shader_face_swap(0); + sgx_shader_srgb_units(0); + sgx_shader_depth_units(0); + sgx_shader_npot_units(0); + sgx_shader_swz_units(NULL, 0); + sgx_shader_cls_units(NULL, 0); + sgx_shader_compile_unlock(); + if (st != SGX_SHADER_OK) { + fprintf(sgx_log(), "sgx: rebuilding for the texel formats " + "failed: %s%s%s\n", sgx_shader_status_name(st), + cerr[0] ? " - " : "", cerr); + return -1; + } + if (!sgx_fs_fits(sh, cerr, sizeof cerr)) { + fprintf(sgx_log(), "sgx: rebuilding for the texel formats no " + "longer fits: %s - the draw is skipped\n", cerr); + return -1; + } + return 0; +} + +static void *sgx_pipe_create_fs(struct pipe_context *pc, + const struct pipe_shader_state *st) +{ + return sgx_pipe_create_shader(pc, st, UIR_STAGE_FRAGMENT); +} + +/* The vertex shader is the draw module's: it is what executes it. The + * hardware still runs the frame's pass-through program, which is what the + * draw module's output is shaped for. */ + +/* The vertex program the part runs has to hand the tiler window coordinates: + * the MTE's own viewport transform is programmed only on the closed driver's + * hardware branch, which needs state this frame does not carry + * (gl-re/hw-tcl-vertex-format.md section 4). So the divide and the viewport + * are appended to the shader here, in TGSI, before it is compiled. + * + * The six viewport floats ride in the vertex record's spare dwords - 8..13, + * which is what is left when two attributes have taken 0..7 - because the + * primary attribute bank has no room for them beside a 4x4 matrix: fourteen + * dwords of record and sixteen of uniform already fill thirty of its + * thirty-two. Those spare dwords are IN[2].xyzw and IN[3].xy. + * + * The way to lift the two-attribute limit is the vertex secondary PDS + * program, which puts constants in the secondary bank and leaves the primary + * bank to the record; work/tcl-capture/ has the descriptors a capture used and + * work/vertex-pds/ section 7 the chunking rule. Its binding word is not in + * this frame, so it is not something this driver can write yet. */ +#define SGX_HWTCL_VP_IN 2 /* first input holding the viewport */ + +static void sgx_vs_src(struct tgsi_src *s, enum tgsi_file file, unsigned index, + unsigned char swizzle) +{ + memset(s, 0, sizeof *s); + s->file = file; + s->index = index; + s->swizzle = swizzle; +} + +/* swizzle selecting one channel in all four positions */ +#define SWZ1(c) (unsigned char)((c) | ((c) << 2) | ((c) << 4) | ((c) << 6)) +#define SWZ_ID 0xe4 + +static int sgx_vs_add_epilogue(struct tgsi_insn *insns, unsigned *n, + unsigned cap, unsigned ntemp, int mte) +{ + unsigned pos = ntemp, rcp = ntemp + 1, i, k; + unsigned char written[TGSI_PARSE_MAX_IO]; + struct tgsi_insn *o; + + if (*n + 6 + TGSI_PARSE_MAX_IO > cap) + return 0; + + /* What the shader leaves unwritten in a varying is not zero, it is + * whatever the output registers held - and the record has fixed slots, + * so an unwritten component is still read. A coordinate's third + * component is the projective divisor and a colour's is a channel; + * one is right for both. */ + memset(written, 0, sizeof written); + for (i = 0; i < *n; i++) + if (insns[i].dst.file == TGSI_F_OUTPUT && + insns[i].dst.index < TGSI_PARSE_MAX_IO) + written[insns[i].dst.index] |= insns[i].dst.writemask; + + /* Everything the shader wrote to the position now lands in a temp, so + * the epilogue has something to read: an output register cannot be + * read back. */ + /* Only when this epilogue transforms: with the MTE doing it, the + * position stays in the output register and its w must survive, since + * the divide happens downstream of emitvtx. */ + if (!mte) + for (i = 0; i < *n; i++) + if (insns[i].dst.file == TGSI_F_OUTPUT && + insns[i].dst.index == 0) + insns[i].dst.file = TGSI_F_TEMPORARY, + insns[i].dst.index = pos; + + /* The END the shader already carries has to stay last. */ + if (*n && insns[*n - 1].opcode == 117) + (*n)--; + else + return 0; + + /* SGX_VS_NODIV drops the projective divide, which is a no-op for any + * draw whose w is one: it tells a defect in the divide apart from one + * in the viewport MADs that follow. */ + if (mte) + goto varyings; + if (SGX_ENVS("SGX_VS_NODIV")) + goto viewport; + o = &insns[(*n)++]; + memset(o, 0, sizeof *o); + o->opcode = 3; /* RCP */ + o->dst.file = TGSI_F_TEMPORARY; + o->dst.index = rcp; + o->dst.writemask = 0xf; + o->nsrc = 1; + sgx_vs_src(&o->src[0], TGSI_F_TEMPORARY, pos, SWZ1(3)); + + o = &insns[(*n)++]; + memset(o, 0, sizeof *o); + o->opcode = 7; /* MUL */ + o->dst.file = TGSI_F_TEMPORARY; + o->dst.index = pos; + o->dst.writemask = 0x7; + o->nsrc = 2; + sgx_vs_src(&o->src[0], TGSI_F_TEMPORARY, pos, SWZ_ID); + sgx_vs_src(&o->src[1], TGSI_F_TEMPORARY, rcp, SWZ1(0)); + +viewport: + /* SGX_VS_NOVP leaves the position where the shader put it instead of + * running it through the viewport, so a caller that already writes + * window coordinates drives the hardware path as a pass-through - + * which is what the software path does, and that renders. */ + if (SGX_ENVS("SGX_VS_NOVP")) { + o = &insns[(*n)++]; + memset(o, 0, sizeof *o); + o->opcode = 1; /* MOV */ + o->dst.file = TGSI_F_OUTPUT; + o->dst.index = 0; + o->dst.writemask = 0x7; + o->nsrc = 1; + sgx_vs_src(&o->src[0], TGSI_F_TEMPORARY, pos, SWZ_ID); + goto wdone; + } + /* x, y and z each scale and offset by their own pair. */ + for (k = 0; k < 3; k++) { + static const unsigned char in[3] = { + SGX_HWTCL_VP_IN, SGX_HWTCL_VP_IN, SGX_HWTCL_VP_IN + 1 + }; + static const unsigned char sc[3] = { 0, 2, 0 }; + static const unsigned char tr[3] = { 1, 3, 1 }; + + o = &insns[(*n)++]; + memset(o, 0, sizeof *o); + o->opcode = 16; /* MAD */ + o->dst.file = TGSI_F_OUTPUT; + o->dst.index = 0; + o->dst.writemask = 1u << k; + o->nsrc = 3; + sgx_vs_src(&o->src[0], TGSI_F_TEMPORARY, pos, SWZ1(k)); + sgx_vs_src(&o->src[1], TGSI_F_INPUT, in[k], SWZ1(sc[k])); + sgx_vs_src(&o->src[2], TGSI_F_INPUT, in[k], SWZ1(tr[k])); + } + +wdone: + /* w is spent - the divide has already happened. */ + o = &insns[(*n)++]; + memset(o, 0, sizeof *o); + o->opcode = 1; /* MOV */ + o->dst.file = TGSI_F_OUTPUT; + o->dst.index = 0; + o->dst.writemask = 0x8; + o->nsrc = 1; + memset(&o->src[0], 0, sizeof o->src[0]); + o->src[0].file = TGSI_F_IMMEDIATE; + o->src[0].swizzle = SWZ_ID; + o->src[0].imm[0] = o->src[0].imm[1] = 1.0f; + o->src[0].imm[2] = o->src[0].imm[3] = 1.0f; + +varyings: + for (i = 1; i < TGSI_PARSE_MAX_IO; i++) { + if (!written[i] || written[i] == 0xf) + continue; + o = &insns[(*n)++]; + memset(o, 0, sizeof *o); + o->opcode = 1; /* MOV */ + o->dst.file = TGSI_F_OUTPUT; + o->dst.index = i; + o->dst.writemask = (unsigned char)(0xf & ~written[i]); + o->nsrc = 1; + o->src[0].file = TGSI_F_IMMEDIATE; + o->src[0].swizzle = SWZ_ID; + o->src[0].imm[0] = o->src[0].imm[1] = 1.0f; + o->src[0].imm[2] = o->src[0].imm[3] = 1.0f; + } + + o = &insns[(*n)++]; + memset(o, 0, sizeof *o); + o->opcode = 117; /* END */ + return 1; +} + +/* Compile the caller's vertex shader for the part. `st` must already be the + * clone taken in sgx_pipe_create_vs(): nir_to_tgsi consumes the NIR it is + * given, and so does the draw module, so the two cannot share one. */ +static struct sgx_shader *sgx_pipe_compile_vs(struct pipe_context *pc, + const struct pipe_shader_state *st) +{ + struct sgx_shader *sh; + + if (!st) + return NULL; + sh = sgx_pipe_create_shader(pc, st, UIR_STAGE_VERTEX); + if (!sh) + return NULL; + /* A program whose uniforms do not fit the primary attribute bank has + * no way to be fed, so it is not a hardware program at all. */ + if (sh->nprimattr > SGX_HWTCL_PA_DEPTH) { + sgx_dbg("vs: %u primary attributes is past the bank\n", + sh->nprimattr); + goto refuse; + } + /* A program the vertex task cannot run stops the core, and a stopped + * core needs the machine power-cycled - so the budget is applied here, + * where the answer is "the draw module runs it", rather than being + * discovered by the part. glxgears' lighting program is the measured + * case: 87 to 95 instructions and 28 temporaries, engaged, and stalled + * on its first frame with no fault reported; the cases that pass use a + * handful of instructions at sixteen temporaries. See + * sgx_hwtcl_max_temps(). */ + if (sh->ntemps > sgx_hwtcl_max_temps()) { + sgx_dbg("vs: %u temporaries is past the %u a vertex task is " + "known to run\n", sh->ntemps, sgx_hwtcl_max_temps()); + goto refuse; + } + if (sh->ninsns > sgx_hwtcl_max_insns()) { + sgx_dbg("vs: %u instructions is past the %u a vertex task is " + "known to run\n", sh->ninsns, sgx_hwtcl_max_insns()); + goto refuse; + } + return sh; +refuse: + sgx_shader_fini(sh); + free(sh); + return NULL; +} + +static void *sgx_pipe_create_vs(struct pipe_context *pc, + const struct pipe_shader_state *st) +{ + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + struct sgx_pipe_vs *v = calloc(1, sizeof(*v)); + struct pipe_shader_state copy, pcopy; + int cloned = 0, pcloned = 0; + + if (!v || !st) { + free(v); + return NULL; + } + /* The clone comes first. Both the draw module and nir_to_tgsi consume + * the NIR they are handed, so compiling from the caller's after the + * draw module has taken it reads a shader that is already gone - + * which is exactly what it did, as a segfault inside nir_to_tgsi. */ + copy = *st; + pcopy = *st; + if (st->type == PIPE_SHADER_IR_NIR && st->ir.nir) { + copy.ir.nir = nir_shader_clone(NULL, st->ir.nir); + cloned = copy.ir.nir != NULL; + /* Both clones before the draw module takes the original, for + * the same reason the first one is taken here: it consumes + * what it is handed, and cloning afterwards read a shader + * that was already gone. */ + pcopy.ir.nir = nir_shader_clone(NULL, st->ir.nir); + pcloned = pcopy.ir.nir != NULL; + /* And one kept whole, to compile a variant from later. Here + * with the others: the draw module below consumes what it is + * handed, so a clone taken afterwards reads a shader that is + * already gone. */ + v->src.ir.nir = nir_shader_clone(NULL, st->ir.nir); + v->have_src = v->src.ir.nir != NULL; + } + v->src.type = st->type; + v->draw_vs = draw_create_vertex_shader(c->draw, st); + if (!v->draw_vs) { + free(v); + return NULL; + } + v->hw = (st->type != PIPE_SHADER_IR_NIR || cloned) ? + sgx_pipe_compile_vs(pc, ©) : NULL; + if (v->hw && (st->type != PIPE_SHADER_IR_NIR || pcloned)) { + sgx_shader_vtx_pack_colour(1); + v->hw_packed = sgx_pipe_compile_vs(pc, &pcopy); + sgx_shader_vtx_pack_colour(0); + } + sgx_dbg("create vs: draw module yes, hardware %s, packed %s\n", + v->hw ? "yes" : "no", v->hw_packed ? "yes" : "no"); + return v; +} + +/* Whether this pair of programs can run the transform on the part. Both binds + * settle it, because the record's width is built into the frame and switching + * it once a frame has content loses the geometry already in it. + * + * A fragment program that iterates anything used to be refused outright, which + * is every textured or Gouraud-shaded draw there is - so the CPU ran the + * vertex stage for all of them. That exclusion existed because the record's + * dwords 8..13 carried the viewport for the shader epilogue; with the MTE + * applying the viewport (the default now) they are the caller's again, and + * what is left to decide is only whether the vertex program has a hardware + * form. It keeps the old rule when SGX_MTE_VP=0 puts the epilogue back. */ +/* The vertex program numbered for the frame this fragment program builds. + * + * Compiled once per pairing and kept; the key changes only when the fragment + * program's sampled inputs, packed colour or set assignment do. */ +static struct sgx_shader *sgx_vs_variant(struct pipe_context *pc, + struct sgx_pipe_vs *v, + const struct sgx_shader *fs) +{ + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + const unsigned char *fsem = NULL, *fidx = NULL; + unsigned char sem[16], idx[16], slot[16]; + unsigned char dw[16]; + struct pipe_shader_state copy; + struct sgx_shader *sh; + unsigned key, n, i, q; + + if (!v->have_src || !fs) + return NULL; + n = sgx_fs_inputs(&c->ctx, &fsem, &fidx); + if (!n || !fsem) + return NULL; + key = ((unsigned)fs->tex_inputs << 8) ^ (unsigned)(fs->packed_in + 1) ^ + ((unsigned)fs->attribs.nset << 24); + for (q = 0; q < fs->attribs.nset && q < XPSB_NSET_MAX; q++) + key ^= (unsigned)fs->set_varying[q] << (4 * q + 16); + for (i = 0; i < v->nvar; i++) + if (v->var_key[i] == key) + return v->hw_var[i]; + if (v->nvar == v->avar) { + unsigned want = v->avar ? v->avar * 2u : 4u; + struct sgx_shader **hv; + unsigned *vk; + + hv = realloc(v->hw_var, want * sizeof(*hv)); + if (!hv) + return NULL; + v->hw_var = hv; + vk = realloc(v->var_key, want * sizeof(*vk)); + if (!vk) + return NULL; + v->var_key = vk; + v->avar = want; + } + + memset(slot, 0, sizeof(slot)); + memset(dw, 0, sizeof(dw)); + for (i = 0; i < n && i < 16; i++) { + int at = -1; + + sem[i] = fsem[i]; + idx[i] = fidx ? fidx[i] : 0; + if (fs->packed_in >= 0 && (int)i == fs->packed_in) { + /* The colour is the second fixed quad of the record. + * Its dword was left unwritten here, so the vertex + * program was told to put it at whatever the stack + * held - which is one of the two reasons a program + * that samples and reads its own attributes could not + * take the part's transform. */ + slot[i] = 1; + dw[i] = SGX_VTX_COLOUR_DW; + continue; + } + for (q = 0; q < fs->attribs.nset && q < XPSB_NSET_MAX; q++) + if (fs->set_varying[q] == (unsigned char)i) { + at = (int)q; + break; + } + slot[i] = (unsigned char)(at < 0 ? 2 : 2 + at); + /* And where that set really starts: the iterator packs them + * by their own widths behind position and colour, so set k + * begins at eight plus the widths before it, not at a quad + * per slot. */ + if (at >= 0 && fs->attribs.set[at].on_colour) { + /* A set carried on a colour iterator lives in the + * colour quad the MTE reads it from, not behind + * the coordinate sets. */ + slot[i] = 1; + dw[i] = (unsigned char) + (fs->attribs.set[at].on_colour == 2u ? + SGX_VTX_COLOUR_DW + 4u : SGX_VTX_COLOUR_DW); + } else if (at >= 0) { + unsigned w = sgx_vtx_set_dw(), q2; + + if (!getenv("SGX_SET_BASE")) + w = xpsb_attrib_set_base(&fs->attribs); + for (q2 = 0; q2 < (unsigned)at && + q2 < XPSB_NSET_MAX; q2++) + if (!fs->attribs.set[q2].on_colour) + w += fs->attribs.set[q2].width; + dw[i] = (unsigned char)w; + } else if (sem[i] == SGX_SEM_COLOR) { + /* A colour the fragment program reads beside a set of + * its own. It belongs in the record's colour quad; + * this used to give it dword 8, which is where set 0 + * begins, so it landed on top of a sampled coordinate + * and both came out wrong. */ + slot[i] = 1; + dw[i] = SGX_VTX_COLOUR_DW; + } else { + /* Neither a coordinate set nor the colour, so nothing + * in the record iterates it and there is no honest + * place to write it. Refuse the variant rather than + * put it somewhere and render whatever that gives. */ + sgx_dbg("vs variant: input %u semantic %u is neither a " + "coordinate set nor the colour\n", i, sem[i]); + return NULL; + } + } + + copy = v->src; + copy.ir.nir = nir_shader_clone(NULL, v->src.ir.nir); + if (!copy.ir.nir) + return NULL; + sgx_shader_vtx_layout_dw(sem, idx, slot, dw, n < 16 ? n : 16); + if (fs->packed_in >= 0) + sgx_shader_vtx_pack_colour(1); + sh = sgx_pipe_compile_vs(pc, ©); + sgx_shader_vtx_pack_colour(0); + sgx_shader_vtx_layout(NULL, NULL, NULL, 0); + if (!sh) + return NULL; + v->hw_var[v->nvar] = sh; + v->var_key[v->nvar] = key; + v->nvar++; + sgx_dbg("vs variant %u: key %08x for %u input(s)\n", v->nvar, key, n); + return sh; +} + +/* How many frames have to open on the part and fall out of it before the + * offer is withdrawn. Low enough that a screen of text does not pay it twice, + * high enough that a frame or two of a caller warming up does not. */ +#define SGX_HWTCL_GIVEUP_MIN 8u + +static int sgx_hwtcl_adapt_off(void) +{ + const char *e = SGX_ENVS("SGX_HWTCL_ADAPT"); + + return e && *e == '0'; +} + +static void sgx_pipe_settle_hwtcl(struct sgx_pipe_context *c) +{ + const struct sgx_shader *fs = c->cur_fs; + const struct sgx_pipe_vs *v = c->vs; + /* nprimattr, not just the coordinate sets: a fragment program that + * reads an interpolated colour has no coordinate set of its own and + * would slip past a test on those alone. */ + /* A program that both samples and reads primary attributes of its own + * cannot take the part's transform. Freeing record dwords 8..13 was + * supposed to make that work and does not: the attributes never reach + * the record and the draw comes out blank. That is the X server's + * composite - nset 1, nprimattr 2, one sampled set - and it is why + * twm's menu had no text. + * + * Both halves are needed in the test. ioquake3's hwtcl geometry has + * nprimattr 0 and is unaffected either way; iterset has eight primary + * attributes and samples nothing, and refusing it as well only moved + * it onto a software path that renders its second set as zero. A + * coordinate set on its own also stays on the part - that is + * ioquake3's textured geometry - and is only kept off it under the + * epilogue scheme. + * + * Retested 2026-09-01 after the vertex program was given the + * iterator's packed offsets, in case that was what this rule was + * really working around. It is not. Lifting it puts every one of + * ioquake3's draws on the part - 12725 against 566 - and costs about + * fifteen per cent less CPU, and the frame comes out wrong: geometry + * blown up across the middle of the screen, which is this failure. + * The frame rate does not move either way, because the scene is bound + * by the render and not by the transform. */ + int sampled = 0; + + if (fs) { + unsigned q; + + for (q = 0; q < fs->attribs.nset && q < XPSB_NSET_MAX; q++) + sampled += fs->attribs.set[q].sampled ? 1 : 0; + } + /* And never a vertex program that samples. The part has no vertex + * texture unit at all - that is why the draw module does the fetch on + * the CPU and why max_texture_samplers is reported from + * SGX_MAX_VTX_SAMPLERS rather than from anything the hardware has. Sent + * to the part anyway, such a program samples texture state nothing + * wrote: the submit comes back -16 and the frame is lost with it. + * sgx_egl_vtxnormal reproduces that in one quad. */ + int vsamp = v && v->hw && v->hw->nsamp; + + int on = c->hwtcl && !c->hwtcl_unprofitable && v && v->hw && !vsamp && + (!(fs && fs->nprimattr && sampled) || SGX_ENVS("SGX_SAMPITER")) && + (sgx_mte_viewport() || !(fs && fs->attribs.nset)); + + if (SGX_ENVS("SGX_DEBUG") && !on) + sgx_dbg("settle hwtcl: off - want %d vs %d hw %d vsamp %d " + "nprimattr %u sampled %d nset %u mte %d\n", c->hwtcl, + v ? 1 : 0, (v && v->hw) ? 1 : 0, vsamp, + fs ? fs->nprimattr : 0u, sampled, + fs ? fs->attribs.nset : 0u, sgx_mte_viewport()); + + /* A fixed-function draw - a fragment program that passes an + * interpolated colour straight out and declares no coordinate set of + * its own - used to be refused here: given to the part it stopped the + * tiling engine mid-scene, about nine frames in ten, with no MMU + * fault, and the kernel recovered the core each time. + * + * That shape carried the colour on the MTE's base slot, iterated as + * the packed dword on USEISSUE_V0. FIX_HW_BRN_25211 is on this + * revision's errata list (hwdefs/sgxerrata.h:785) and its whole + * workaround is to stop using that slot: the vendor routes the vertex + * colour through a texture coordinate set instead + * (opengles1/fftnlgles.c:900-975, usegles.c:2077-2118). The fragment + * program can be compiled that way under SGX_FF_COLOUR_SET=1 - see + * sgx_shader.c's ff_colour_set - but not by default: the routing + * loses the triangle, the five packed-colour cases and the line strip + * on the part, so it costs more than the stall it avoids until that + * is understood. driver/test/sgx_ff_quads.c is the twenty-second + * reproducer of the stall itself. */ + if (on && fs && fs->colour_passthrough && !fs->attribs.nset && + !SGX_ENVS("SGX_FF_COLOUR_HWTCL")) { + sgx_dbg("settle hwtcl: a fixed-function colour pass-through " + "stops the part\n"); + on = 0; + } + + sgx_pipe_settle_flatshade(c); + + /* Not while anything is clipped. The scissor is enforced by cutting + * the triangles up in sgx_vbuf_emit(), which is the draw module's + * path: on the part's own transform nothing cuts them and the clip is + * simply not applied. Measured against a server with no acceleration + * at all, every clipped draw came out unclipped - a rectangle clip + * filled 1600 pixels where 576 were asked for, and a bitmap clip 1600 + * where 208 were - while all nineteen other requests matched exactly. + * That is a window manager's buttons drawn as solid blocks and the + * black bar across the top of a Window Maker screen. + * + * Refused rather than fixed here: the honest fix is a scissor the + * hardware applies, and this is the same rule sgx_vbuf_native_prim() + * already follows for the same reason. SGX_HWTCL_SCISSOR keeps the + * transform on, which is what this did before. */ + /* gl_FragCoord is fed from the record's position slot + * (sgx_vbuf_layout: the set is emitted from the position), which + * holds window coordinates only when the draw module put them there. + * Under the part's transform the slot is the program's clip-space + * output and the MTE's viewport never reaches the iterated set, so + * such a program stays on the draw module. */ + if (on && fs && fs->fragcoord_set >= 0) { + sgx_dbg("settle hwtcl: the program reads gl_FragCoord, which " + "only the draw module's record carries\n"); + on = 0; + } + + if (on && sgx_scissor_clips(&c->ctx) && + !SGX_ENVS("SGX_HWTCL_SCISSOR")) { + sgx_dbg("settle hwtcl: a scissor is in force and only the " + "draw module applies one\n"); + on = 0; + } + + /* Where the two stages disagree about a second coordinate set. The + * vertex program writes output slot k at emitted dword 4*k - position, + * colour, then a set per slot - while the iterator packs the sets by + * their own widths behind the eight fixed dwords. They agree on the + * first set, and on any set that is four floats wide; a three-float + * set 0 puts everything after it two dwords out, and the second set + * would be iterated from the wrong place. */ + /* The two stages used to disagree about a second coordinate set: the + * vertex program wrote output slot k at emitted dword 4*k while the + * iterator packs the sets by their own widths behind the eight fixed + * dwords, so anything but a four-float set 0 put the sets after it two + * dwords out. The vertex program is now told the packed offset - see + * sgx_shader_vtx_layout_dw() - so they agree for any width and this + * no longer has to refuse the transform. SGX_NO_PACKED_VSOUT puts the + * old refusal back. */ + if (on && fs && fs->attribs.nset > 1 && fs->attribs.set[0].width != 4 && + SGX_ENVS("SGX_NO_PACKED_VSOUT")) { + sgx_dbg("settle hwtcl: %u coordinate set(s), the first %u " + "float(s) wide - the second would not line up with " + "the vertex program's slots\n", fs->attribs.nset, + (unsigned)fs->attribs.set[0].width); + on = 0; + } + + c->hwtcl_ok = on; + /* The record's width is built into the frame, so a switch has to + * happen at a frame boundary: doing it with geometry in the frame + * rebuilds the frame and throws that geometry away. Which is reachable + * now that most programs qualify - a scene binding one program the + * part transforms and one it does not lands here on every bind. */ + /* Only one direction has to end the frame. Going from the part's + * transform to the draw module's does: the record the frame has + * already built is fourteen dwords and the software path emits + * eleven, so the geometry in it would be read at the wrong stride. + * Going the other way does not - the draw module can emit any + * program, including one the part would have taken - so a frame that + * has started in software finishes in software and the switch waits + * for the next one. + * + * ioquake3 binds programs the part takes and programs it does not + * throughout the frame, and used to split on every change in either + * direction. SGX_SPLIT_BOTH_WAYS restores that. */ + /* A frame may hold both kinds - a record carries its own vertex width + * and its own vertex PDS copy, and the copy's program slot is the + * record's answer rather than the frame's; before that last part a + * mixed frame rendered a texture atlas instead of the scene. + * + * It is still not worth doing. Taking the part's transform mid-frame + * turns the vertex-program split back on for the rest of that frame - + * that split is guarded on the frame's own hwtcl - and ioquake3 pays + * two extra submits a swap for a handful of draws the part is slower + * at anyway: 3.0 submits a swap and 21 fps against 1.0 and 23, with + * the whole difference in flush and draw. So the frame's transform is + * settled by the draw that opened it. SGX_MIX_TRANSFORM allows both + * kinds in one frame again. */ + if (on && !c->ctx.hwtcl && c->ctx.draws && + !SGX_ENVS("SGX_MIX_TRANSFORM") && + !SGX_ENVS("SGX_SPLIT_BOTH_WAYS")) { + sgx_dbg("settle hwtcl: staying on the draw module for the " + "rest of the frame rather than splitting it\n"); + on = 0; + c->hwtcl_ok = 0; + } + if (on != c->ctx.hwtcl && c->ctx.draws && + !SGX_ENVS("SGX_MIX_TRANSFORM")) { + sgx_dbg("settle hwtcl: %d -> %d with %u draw(s) in the frame; " + "sending them first\n", c->ctx.hwtcl, on, + c->ctx.draws); + sgx_perf_note("transform path"); + /* The frame opened on the part and could not stay there. + * Counted rather than only paid: see below. */ + c->hwtcl_fallbacks++; + sgx_flush(&c->ctx); + sgx_set_clear_enable(&c->ctx, 0); + } + /* A caller that keeps opening frames on the part and falling out of + * them is paying a submit each time for a transform it does not get + * to keep. The X server is the case: its shaders sample and read + * primary attributes of their own, which the part cannot transform, + * but enough of its draws qualify to open a frame on it - a hundred + * and twenty-two early submits painting one screen of text, which is + * most of the two seconds that took. Measured rather than assumed, + * and only once there is enough of a history to mean anything. + * + * SGX_HWTCL_ADAPT=0 keeps offering it however badly it does. */ + if (!c->hwtcl_unprofitable && + c->hwtcl_fallbacks >= SGX_HWTCL_GIVEUP_MIN && + !sgx_hwtcl_adapt_off()) { + c->hwtcl_unprofitable = 1; + sgx_dbg("settle hwtcl: %u frame(s) opened on the part had to " + "be sent early; staying on the draw module\n", + c->hwtcl_fallbacks); + } + sgx_set_hwtcl(&c->ctx, on); +} + +static void sgx_pipe_bind_fs(struct pipe_context *pc, void *state) +{ + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + + sgx_dbg("bind fs: %s\n", state ? "a compiled program" : "none"); + /* A frame carries one attribute configuration, and it is built from + * whatever program is bound when the frame is flushed. So a program + * whose varyings are laid out differently ends the frame rather than + * joining it: alacritty draws its glyphs and then a rectangle whose + * program has one input, and the glyph records were iterated with that + * one - their packed colour never arrived, and the background colour + * they read out of the stale register painted a blue rectangle behind + * every run of text. */ + if (c->ctx.draws && state && c->cur_fs && state != c->cur_fs && + !SGX_ENVS("SGX_NO_ATTRIB_SPLIT")) { + const struct sgx_shader *ns = state, *os = c->cur_fs; + + /* The frame ends on any program change, not only on a layout + * change. A frame also carries one fragment + * literal pool - ctx_upload_pools() writes the bound program's + * into the single constant block - and one iterator + * configuration, which ctx_emit_samplers() builds from the + * first record's program, so two programs sharing a frame is + * a real hazard. + * + * A recorded negative, and a warning about how noisy this + * measurement is: on ioquake3's ammo counter it looked like + * the fix - 6 of 8 frames correct against 0 of 32 for four + * other configurations - and two fresh runs with the same + * rule made default were 0 of 24. That stays a caution about + * the measurement, not a verdict on the rule. */ + /* Any change of program, not only a change of attribute + * layout. The reasoning above is the whole argument for it: + * the frame carries one fragment literal pool and one + * iterator configuration, both built from a single program, + * so a second program in the same frame runs against the + * first one's. Comparing layouts caught only the cases where + * that difference was visible in the attribute sets. + * + * glmark2's ideas scene is what settled it. Its page, its text + * and its background gradient all rendered black, and every + * one of them came back when the frame was split on each + * program change - while comparing tex_inputs and + * set_varying[] as well as the sets, which is everything + * sgx_vs_variant() keys on, changed nothing. So the programs + * that collide need not differ in their layout at all. + * + * The ideas logo settled the rest of it: its own program's + * output had no effect on the pixels at all - a literal, a + * different literal and a uniform all rendered pure white - + * because the record ran the program the frame had installed. + * Splitting on every program change is what makes a draw run + * its own program, so it is the default. + * + * SGX_NO_FS_SPLIT keeps the old layout-only comparison. */ + /* Narrowed to the one thing a frame really cannot carry two + * of: the width of the record the vertex fetch walks. + * + * Splitting on every program change - and before that on any + * difference in the sets - both predate per-record attributes. + * ctx_emit_draw_records() now writes each record's own + * group-14 word, group-10 width and primary PDS pointer into + * its own state window before copying it, so two programs that + * describe their varyings differently no longer read each + * other's iterator. What is still frame-global is the vertex + * fetch: its DMA control and byte stride sit at heap 0x3a4 and + * 0x3c0, outside that window, so one frame names one width. + * + * Every split is a whole frame - its own setup, its own tiler + * pass and a render that loads back and stores every tile - so + * this is the difference between ioquake3 costing ten submits + * a swap and seven. SGX_FS_SPLIT_ALL restores the old rule. + */ + /* The width no longer ends the frame: each record carries its + * own vertex PDS program copy, and the fetch's DMA control and + * byte stride are that program's data. SGX_WIDTH_SPLIT and + * SGX_FS_SPLIT_ALL put the old rule back. + * + * The packed-input half is gone with it. Counted apart over 64 + * swaps of ioquake3 it fired zero times: it tests something + * ctx_emit_draw_records() has written per record since the + * group-14 work, so it cannot fire. SGX_PACKED_SPLIT restores + * it for anyone who finds a case that disagrees. */ + int width_split = (SGX_ENVS("SGX_WIDTH_SPLIT") || + SGX_ENVS("SGX_NO_VPDS_COPY")) && + xpsb_attrib_stride(&ns->attribs) != + xpsb_attrib_stride(&os->attribs); + + if (SGX_ENVS("SGX_FS_SPLIT_ALL") || width_split || + (SGX_ENVS("SGX_PACKED_SPLIT") && + ns->packed_in != os->packed_in)) { + unsigned nd = c->ctx.draws; + int fret; + + sgx_perf_note(width_split ? "record width" : + "packed input"); + fret = sgx_flush(&c->ctx); + /* Pixmap frames leave through here, not through the + * framebuffer switch, so this is where they can be + * looked at. Same gate: SGX_DUMP_RT plus a width. */ + { + const char *dp = SGX_ENVS("SGX_DUMP_RT"); + const char *dw = SGX_ENVS("SGX_DUMP_RT_W"); + + if (dp && dw && *dw && !fret && c->ctx.fb.width && + c->ctx.fb.width <= (unsigned)atoi(dw)) { + static unsigned dseq; + char dn[256]; + + snprintf(dn, sizeof dn, "%s-fs%u.ppm", + dp, dseq++); + sgx_dump_target(&c->ctx, dn); + } + } + + sgx_dbg("bind fs: sent %u draw(s) first; the frame " + "cannot carry both layouts: %d\n", nd, fret); + sgx_set_clear_enable(&c->ctx, 0); + } + } + c->cur_fs = state; + sgx_bind_fs(&c->ctx, (const struct sgx_shader *)state); + sgx_pipe_settle_hwtcl(c); +} + +/* Bind whichever vertex program the transform path now calls for. Split out of + * sgx_pipe_bind_vs() because the choice depends on c->hwtcl, which the re-arm + * in sgx_pipe_draw_vbo() can change *after* the bind has already happened: a + * frame that fell back to the draw module leaves the pass-through bound, and + * the next shader was bound while the flag was still down, so re-arming the + * flag alone left the part running a program with no instructions and the draw + * came out empty. That is the "uniform mat4" case - it passes on its own and + * fails after any test that falls back. */ +static void sgx_pipe_apply_vs(struct sgx_pipe_context *c) +{ + struct pipe_context *pc = &c->base; + struct sgx_pipe_vs *v = c->vs; + + /* Under hardware transform the part runs the caller's own program; + * otherwise it runs the frame's pass-through, because what reaches it + * has already been through the draw module. The context refuses a draw + * without one or the other. */ + if (c->hwtcl && v && v->hw) { + const struct sgx_shader *fs = c->cur_fs; + /* Numbered for this pairing where one can be built; the forms + * compiled at create time are the fallback. */ + struct sgx_shader *pick = sgx_vs_variant(pc, v, fs); + + if (!pick) + pick = (fs && fs->packed_in >= 0 && v->hw_packed) ? + v->hw_packed : v->hw; + /* What the part will actually run, against what it is running. + * The frame ends on a change of pipe object, but two objects + * can compile to the same code - which is the whole question + * of whether the split is needed as often as it fires. */ + /* The frame ends here, not only in bind_vs. The program the + * part actually runs is picked from the *pairing* - a + * different fragment program can select a different vertex + * variant with no bind_vs between - and a frame carries one + * vertex entry point, patched frame-wide out of the + * relocation table. Watching the pipe object alone let two + * programs share a frame whenever the pairing changed, and + * what came out was the second program's transform over the + * first program's geometry. + * + * Behind SGX_VS_PAIR_SPLIT and NOT yet shown to work: the run + * that first carried it took the machine down hard - no ping, + * a power cycle to recover - and whether that was this or + * something the run happened to reach is unestablished. A + * flush per pairing change may simply be far more submits than + * the part will take. Reproduce on a machine you can power + * cycle, and watch the submit count before the frame rate. */ + if (SGX_ENVS("SGX_VS_PAIR_SPLIT") && + c->ctx.draws && pick && c->ctx.vs && pick != c->ctx.vs && + c->ctx.hwtcl && !SGX_ENVS("SGX_NO_VS_SPLIT")) { + unsigned nd = c->ctx.draws; + + sgx_perf_note("vertex program"); + sgx_flush(&c->ctx); + if (nd) + sgx_set_clear_enable(&c->ctx, 0); + sgx_dbg("apply vs: sent %u draw(s) first; the frame " + "carries one vertex program\n", nd); + } + if (SGX_ENVS("SGX_VS_IDENT")) { + uint32_t h = 2166136261u; + unsigned q; + + for (q = 0; pick && pick->code && q < pick->ninsns * 2u; + q++) + h = (h ^ ((const uint32_t *)pick->code)[q]) * + 16777619u; + fprintf(sgx_log(), "sgx: vs ident: %p code %08x " + "insns %u\n", (const void *)pick, h, + pick ? pick->ninsns : 0u); + } + sgx_bind_vs(&c->ctx, pick); + } + else + sgx_bind_vs(&c->ctx, sgx_passthrough_vs()); + sgx_dbg("bind vs: %s\n", (c->hwtcl && v && v->hw) ? "on the part" : + "the draw module's"); +} + +static void sgx_pipe_bind_vs(struct pipe_context *pc, void *state) +{ + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + struct sgx_pipe_vs *v = state; + + /* A vertex program change ends the frame too, when one is open. The + * record layout a frame is built for comes from the pairing, so two + * vertex programs in one frame have the same hazard two fragment + * programs do - and every rule guarding that was on the fragment side + * only, so a scene that changes vertex program with the fragment one + * held was never covered. SGX_NO_VS_SPLIT turns it off. */ + /* Only when the part is running the transform. The hazard this guards + * is the record layout, and the vertex program writes that only on the + * hardware path; on the draw module's path the record comes from the + * fragment program's attributes, so two vertex programs in one frame + * describe the same record and there is nothing to split for. + * + * Splitting anyway cost glxgears two of its three gears: fixed + * function makes a vertex program per material, so each gear ended the + * frame, and only the first survived to be seen. Off the hardware path + * it renders all three at 60 fps against 39. */ + if (c->ctx.draws && state && c->vs && state != c->vs && c->ctx.hwtcl && + !SGX_ENVS("SGX_NO_VS_SPLIT")) { + unsigned nd = c->ctx.draws; + + sgx_perf_note("vertex program"); + sgx_flush(&c->ctx); + if (nd) + sgx_set_clear_enable(&c->ctx, 0); + sgx_dbg("bind vs: sent %u draw(s) first; the frame carries one " + "record layout\n", nd); + } + c->vs = state; + draw_bind_vertex_shader(c->draw, v ? v->draw_vs : NULL); + /* The frame is built for one record width, so whether the part runs + * the transform is settled here, before any draw. A shader with no + * hardware program puts the frame back to the narrow record. */ + sgx_pipe_settle_hwtcl(c); + sgx_pipe_apply_vs(c); +} + +static void sgx_pipe_delete_vs(struct pipe_context *pc, void *state) +{ + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + struct sgx_pipe_vs *v = state; + unsigned i; + + if (!v) + return; + if (c->vs == state) + c->vs = NULL; + /* ctx->vs points at v->hw, not at v, so clearing c->vs is not enough: + * the upload path would still read the freed hardware shader. */ + sgx_forget_vs(&c->ctx, v->hw); + sgx_forget_vs(&c->ctx, v->hw_packed); + for (i = 0; i < v->nvar; i++) + sgx_forget_vs(&c->ctx, v->hw_var[i]); + draw_delete_vertex_shader(c->draw, v->draw_vs); + sgx_shader_fini(v->hw); + free(v->hw); + sgx_shader_fini(v->hw_packed); + free(v->hw_packed); + for (i = 0; i < v->nvar; i++) { + sgx_shader_fini(v->hw_var[i]); + free(v->hw_var[i]); + } + free(v->hw_var); + free(v->var_key); + if (v->have_src && v->src.ir.nir) + ralloc_free(v->src.ir.nir); + free(v); +} + +/* pipe_context::set_vertex_buffers and the vertex element state. + * + * Gallium describes a vertex any way the application likes; the hardware's + * record is eleven floats. The conversion is by element index: the first is + * the position, the second the colour, the third a coordinate set. */ +/* Float vertex data, at 32 or 16 bits a component, and the component count + * with it. *half, when given, says which. + * + * Half floats are a vertex format this part reads without help: the fetch is + * a byte DMA that rounds only to whole dwords (codegen/pds/pds.c:1730, + * "Size is in bytes - round up to nearest 32 bit word") and carries no format + * field at all (EURASIA_PDS_DOUTD0/1, sgxdefs.h:4212-4258 - a source address, + * a burst size, a line count, an attribute offset and a stride), and the + * expansion is one UNPCKF32F16 a component, which the vendor emits for + * exactly this case (opengles2/use.c:1512-1530, GLES2_STREAMTYPE_HALFFLOAT -> + * EURASIA_USE1_PCK_FMT_F16). */ +/* Takes either a whole enum pipe_format or the low byte a + * pipe_vertex_element carries: every format named here is below 256, so the + * two are the same value. */ +/* What the part's vertex fetch can be handed, which is wider than what the + * draw module ever emits. The draw module transforms on the CPU and hands over + * floats, so sgx_pipe_float_comps_fmt() below stays float-only and the software + * path is untouched; this one is the transform path's, and it also takes the + * normalised byte colour every GL application sends. Refusing that was what + * kept ioquake3 off the part: one lightmapped draw a frame carries a + * four-byte colour, and the first draw that cannot go to the part takes the + * rest of the frame back to the draw module with it. */ +static unsigned sgx_pipe_attr_comps_fmt(unsigned f, int *half, int *u8n, + int *bgra) +{ + unsigned n = sgx_pipe_float_comps_fmt(f, half); + + if (u8n) + *u8n = 0; + if (bgra) + *bgra = 0; + if (n) + return n; + if (f == PIPE_FORMAT_R8G8B8A8_UNORM || f == PIPE_FORMAT_B8G8R8A8_UNORM) { + if (half) + *half = 0; + if (u8n) + *u8n = 1; + if (bgra) + *bgra = f == PIPE_FORMAT_B8G8R8A8_UNORM; + return 4; + } + return 0; +} + +static unsigned sgx_pipe_float_comps_fmt(unsigned f, int *half) +{ + unsigned n = 0, h = 0; + + if (f == PIPE_FORMAT_R32_FLOAT) n = 1; + else if (f == PIPE_FORMAT_R32G32_FLOAT) n = 2; + else if (f == PIPE_FORMAT_R32G32B32_FLOAT) n = 3; + else if (f == PIPE_FORMAT_R32G32B32A32_FLOAT) n = 4; + else if (f == PIPE_FORMAT_R16_FLOAT) { n = 1; h = 1; } + else if (f == PIPE_FORMAT_R16G16_FLOAT) { n = 2; h = 1; } + else if (f == PIPE_FORMAT_R16G16B16_FLOAT) { n = 3; h = 1; } + else if (f == PIPE_FORMAT_R16G16B16A16_FLOAT) { n = 4; h = 1; } + if (half) + *half = (int)h; + return n; +} + +static unsigned sgx_pipe_float_comps(uint8_t f) +{ + return sgx_pipe_float_comps_fmt(f, NULL); +} + +/* The vertex elements and buffers go to the draw module, which is what reads + * them: it runs the vertex shader, and the hardware never sees the caller's + * layout at all. */ +static void *sgx_pipe_create_velems(struct pipe_context *pc, unsigned n, + const struct pipe_vertex_element *el) +{ + struct sgx_pipe_velems *v; + + (void)pc; + if (!el || n > PIPE_MAX_ATTRIBS) + return NULL; + v = calloc(1, sizeof(*v)); + if (!v) + return NULL; + memcpy(v->el, el, n * sizeof(*el)); + v->n = n; + return v; +} + +static void sgx_pipe_bind_velems(struct pipe_context *pc, void *state) +{ + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + struct sgx_pipe_velems *v = state; + + c->velems = v; + draw_set_vertex_elements(c->draw, v ? v->n : 0, v ? v->el : NULL); +} + +static void sgx_pipe_delete_velems(struct pipe_context *pc, void *state) +{ + (void)pc; + free(state); +} + +static void sgx_pipe_set_vertex_buffers(struct pipe_context *pc, unsigned n, + const struct pipe_vertex_buffer *vb) +{ + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + unsigned i; + + if (n > PIPE_MAX_ATTRIBS) + return; + for (i = 0; i < PIPE_MAX_ATTRIBS; i++) + c->vb[i] = (i < n && vb) ? vb[i] : + (struct pipe_vertex_buffer){ 0 }; + c->nvb = n; + draw_set_vertex_buffers(c->draw, n, vb); +} + +/* The draw module reads vertices and indices with the CPU, so a resource has + * to be mapped for it. The mapping is left in place: a resource is mapped once + * and drawn from many times, and the winsys counts the nesting. */ +static const void *sgx_pipe_resource_cpu(struct pipe_context *pc, + struct pipe_resource *pres) +{ + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + struct sgx_pipe_resource *r = (struct sgx_pipe_resource *)pres; + void *p = NULL; + + if (!r || sgx_resource_map(&r->r, c->ctx.ws, &p) != SGX_RESOURCE_OK) + return NULL; + return p; +} + +/* The same contents, in cached memory. + * + * The object is mapped write-combining and a CPU read from that is an order of + * magnitude slower than a cached one. sgx_hwtcl_draw() reads the caller's + * vertex data once per triangle corner per attribute, which for a seven + * thousand triangle model is tens of thousands of small scattered reads a + * frame - fifty-six per cent of the process's CPU time on glmark2's shading + * scene, which spends eight milliseconds of its seventy on the part. One bulk + * sequential read, which write-combining memory is good at, serves them all. + * + * Only for buffers: a texture is twiddled and has sgx_resource_map_level() for + * this. Falls back to the mapping itself if the copy cannot be made, so a + * failure here costs speed and not correctness. */ +static const void *sgx_pipe_resource_cpu_cached(struct pipe_context *pc, + struct pipe_resource *pres) +{ + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + struct sgx_pipe_resource *r = (struct sgx_pipe_resource *)pres; + const void *p = sgx_pipe_resource_cpu(pc, pres); + + if (!p || !r) + return p; + if (pres->target != PIPE_BUFFER || SGX_ENVS("SGX_NO_VTX_SHADOW")) + return p; + /* Not while someone can be writing it. The copy is refreshed when a + * write is announced, and a mapping that stays open announces nothing + * - which is how xsetroot's fill reached the part as four vertices at + * the origin, and drew nothing. */ + if (r->write_mapped) { + r->r.shadow_stale = 1; + return p; + } + { + /* What the object holds, not what the resource declares: a + * copy of width0 out of a smaller mapping reads past its end. */ + uint32_t n = pres->width0 < r->r.size ? pres->width0 : + (uint32_t)r->r.size; + + if (!n) + return p; + if (!r->r.shadow || r->r.shadow_size < n) { + void *b = realloc(r->r.shadow, n); + + if (!b) + return p; + r->r.shadow = b; + r->r.shadow_size = n; + r->r.shadow_stale = 1; + } + if (r->r.shadow_stale) { + memcpy(r->r.shadow, p, n); + r->r.shadow_stale = 0; + } else if (SGX_ENVS("SGX_VTX_SHADOW_CHECK") && + memcmp(r->r.shadow, p, n)) { + uint32_t q, bad = 0, first = n; + + for (q = 0; q < n; q++) + if (((const char *)r->r.shadow)[q] != + ((const char *)p)[q]) { + if (first == n) first = q; + bad++; + } + fprintf(sgx_log(), "sgx: vertex shadow stale: %u of " + "%u byte(s) differ from offset %u, resource " + "%p\n", bad, n, first, (void *)pres); + } + } + (void)c; + return r->r.shadow; +} + +/* Anything that writes a buffer has to say so, or the next draw reads the copy + * taken before the write. */ +static void sgx_pipe_shadow_dirty(struct pipe_resource *pres) +{ + struct sgx_pipe_resource *r = (struct sgx_pipe_resource *)pres; + + if (r) + r->r.shadow_stale = 1; +} + + +/* ---- hardware transform ------------------------------------------------ + * + * When the vertex program runs on the part, the draw module is out of the + * path entirely: what the hardware fetches is the caller's own vertex data, + * and the program reads it as primary attributes. The compiler puts input i + * at pa[4*i], so the record is the caller's attributes at four-dword strides + * - and the record is fourteen dwords, so there is room for three of them. + * + * The hardware's draw record is a triangle list, so anything else has to be + * expanded into one here. That is index arithmetic, not vertex processing: + * no attribute is read, computed or transformed on the CPU. */ +/* How many of the caller's vertex attributes the part's own transform path + * will take. The compiler puts input i at pa[4*i] and sgx_hwtcl_draw() writes + * it at record dword 4*i, so the record's width is the whole limit: + * + * - with the shader epilogue applying the viewport it rides in dwords + * 8..13, so two attributes fit; + * - with the MTE applying it those dwords are the caller's, so three fit in + * the fourteen-dword record - a position, a texture coordinate and a + * colour, which is what a fixed-function draw is made of. + * + * Two is what the suite was measured at (56 of 63, no faults, 2026-08-28); + * three is the same record with the dwords the viewport vacated used for what + * they were always meant to carry. A fourth attribute needs a sixteen-dword + * record, which nothing has run yet. SGX_HWTCL_MAX_IN pins it either way. + * + * One was the value before, and it is why almost nothing used this path: any + * real draw has a position and a colour or a coordinate, so it fell back to + * interpreting the vertex program on the CPU, which is where a profile of + * ioquake3 puts a sixth of the time. */ +#define SGX_HWTCL_MAX_IN sgx_hwtcl_max_in() + +static unsigned sgx_hwtcl_max_in(void) +{ + static int n = -1; + + if (n < 0) { + const char *e = SGX_ENVS("SGX_HWTCL_MAX_IN"); + unsigned room = sgx_hwtcl_record_dwords(0) / 4u; + + n = e && *e ? atoi(e) : (sgx_mte_viewport() ? 3 : 2); + if (n < 1) + n = 1; + /* An attribute past the record's end would be written outside + * it and read as another attribute's dwords. */ + if ((unsigned)n > room) + n = (int)room; + } + return (unsigned)n; +} + +static int sgx_hwtcl_prim_ok(unsigned mode) +{ + switch (mode) { + case MESA_PRIM_TRIANGLES: + case MESA_PRIM_TRIANGLE_STRIP: + case MESA_PRIM_TRIANGLE_FAN: + case MESA_PRIM_QUADS: + case MESA_PRIM_QUAD_STRIP: + return 1; + default: + return 0; + } +} + +/* How many triangle corners `n` vertices of this mode produce. */ +static unsigned sgx_hwtcl_tri_verts(unsigned mode, unsigned n) +{ + switch (mode) { + case MESA_PRIM_TRIANGLES: return (n / 3) * 3; + case MESA_PRIM_TRIANGLE_STRIP: + case MESA_PRIM_TRIANGLE_FAN: return n >= 3 ? (n - 2) * 3 : 0; + case MESA_PRIM_QUADS: return (n / 4) * 6; + case MESA_PRIM_QUAD_STRIP: return n >= 4 ? ((n - 2) / 2) * 6 : 0; + default: return 0; + } +} + +/* The corner-th corner of the tri-th triangle, as an offset into the draw. + * + * The MTE's flat-shade field names the corner the colour is taken from - + * the last by default, the first under flatshade_first (sgx_raster_word()) + * - so the decomposition puts GL's provoking vertex at that corner. A + * rotation of the three corners keeps the winding. */ +static unsigned sgx_hwtcl_corner(unsigned mode, unsigned tri, unsigned corner, + unsigned first, unsigned flat) +{ + switch (mode) { + case MESA_PRIM_TRIANGLES: + return tri * 3 + corner; + case MESA_PRIM_TRIANGLE_STRIP: + /* Odd triangles wind the other way, so two corners swap; the + * provoking vertex is tri + 2, or tri under the first + * convention. */ + if (tri & 1) + return first ? + tri + (corner == 0 ? 0 : corner == 1 ? 2 : 1) : + tri + (corner == 0 ? 1 : corner == 1 ? 0 : 2); + return tri + corner; + case MESA_PRIM_TRIANGLE_FAN: + /* Provoking vertex tri + 2, or tri + 1 under the first one. */ + if (first) + return corner == 2 ? 0 : tri + 1 + corner; + return corner == 0 ? 0 : tri + corner; + case MESA_PRIM_QUADS: { + /* A quad's provoking vertex is its fourth, and only the other + * diagonal has it in both triangles - taken for a flat-shaded + * draw alone, so a smooth one keeps the split it had. */ + static const unsigned c[6] = { 0, 1, 2, 0, 2, 3 }; + static const unsigned cl[6] = { 0, 1, 3, 1, 2, 3 }; + + return (tri / 2) * 4 + + (flat && !first ? cl : c)[(tri & 1) * 3 + corner]; + } + case MESA_PRIM_QUAD_STRIP: { + static const unsigned c[6] = { 0, 1, 3, 0, 3, 2 }; + static const unsigned cl[6] = { 0, 1, 3, 2, 0, 3 }; + + return (tri / 2) * 2 + (first ? c : cl)[(tri & 1) * 3 + corner]; + } + default: + return 0; + } +} + +/* Whether this draw can go straight to the part. */ +static int sgx_hwtcl_can(struct sgx_pipe_context *c, + const struct pipe_draw_info *info) +{ + const struct sgx_pipe_vs *v = c->vs; + unsigned i; + + if (!c->hwtcl || !c->hwtcl_ok || !v || !v->hw || !c->velems || + !c->nvb) { + /* Settled at bind time, so every draw after the first reports + * the same thing - and each term separately, because "refused" + * used to name only the one reason that no longer applies. */ + if (c->hwtcl) + sgx_dbg("hwtcl: refused - hwtcl %d ok %d vs %d hw %d " + "velems %d nvb %u%s\n", + c->hwtcl, c->hwtcl_ok, v != NULL, + (v && v->hw) ? 1 : 0, c->velems != NULL, + c->nvb, + (v && !v->hw) ? " (the vertex program has no " + "hardware form: see create vs)" : + (!c->hwtcl_ok && !sgx_mte_viewport()) ? + " (a fragment program with any varying cannot " + "use the part's transform while the record's " + "slots 8..13 hold the viewport; SGX_MTE_VP)" : + ""); + return 0; + } + if (info->instance_count > 1 || info->index_size > 4) { + sgx_dbg("hwtcl: refused - %u instance(s), index size %u\n", + info->instance_count, info->index_size); + return 0; + } + /* Every element in a form the record's fetch can read. The screen + * offers more than this for a vertex buffer because the draw module + * reads them, and a format resolved to no components here would be + * fetched as nothing at all. */ + for (i = 0; i < c->velems->n; i++) + if (!sgx_pipe_attr_comps_fmt((unsigned)c->velems->el[i].src_format, + NULL, NULL, NULL)) { + sgx_dbg("hwtcl: refused - attribute %u is format %u, " + "which the fetch cannot take\n", i, + (unsigned)c->velems->el[i].src_format); + return 0; + } + /* A restart index is a value inside the index buffer, not a count, so + * the fetch would take 0xffff or 0xffffffff for a vertex number and + * read far outside the buffer - a bad submit, which on this part is a + * lock that needs the power button. The draw module honours it. */ + if (info->primitive_restart) { + sgx_dbg("hwtcl: refused - primitive restart\n"); + return 0; + } + if (!sgx_hwtcl_prim_ok(info->mode)) { + sgx_dbg("hwtcl: refused - primitive mode %u\n", info->mode); + return 0; + } + /* State the draw module applies in its own pipeline + * (draw_need_pipeline()) and the part's transform has no path for: + * an unfilled polygon mode, two-sided lighting and the user clip + * planes. The MTE has clip planes, but the state block carries no + * cull-plane group yet, so the draw that needs them goes where it is + * rendered right. A filled polygon's depth bias is ISP word B's + * (sgx_isp_word1()); a line's or a point's is still widened on the + * CPU with the primitive. */ + if (c->cur_rast) { + const struct pipe_rasterizer_state *r = + &((const struct sgx_pipe_rast *)c->cur_rast)->pipe; + + if (r->fill_front != PIPE_POLYGON_MODE_FILL || + r->fill_back != PIPE_POLYGON_MODE_FILL || + r->offset_line || r->offset_point || + r->light_twoside || r->clip_plane_enable || + r->poly_stipple_enable) { + sgx_dbg("hwtcl: refused - polygon mode %u/%u, line/" + "point offset %u/%u, two-sided %u, clip planes " + "0x%x, stipple %u are the draw module's\n", + r->fill_front, r->fill_back, r->offset_line, + r->offset_point, r->light_twoside, + r->clip_plane_enable, r->poly_stipple_enable); + return 0; + } + } + /* The record is fourteen dwords and attribute i sits at 4*i, so what + * fits depends on whether the viewport still rides in dwords 8..13. + * See sgx_hwtcl_max_in() and sgx_vs_add_epilogue(). */ + if (c->velems->n > SGX_HWTCL_MAX_IN) { + sgx_dbg("hwtcl: refused, %u attribute(s) and the record takes " + "%u\n", c->velems->n, SGX_HWTCL_MAX_IN); + return 0; + } + + sgx_dbg("hwtcl: %u attribute(s), %u vertex constant(s)\n", + c->velems->n, c->vs_const_n); + if (c->vs_const_n > (sgx_hwtcl_limm_uniforms() ? + SGX_HWTCL_MAX_UNI_LIMM : SGX_HWTCL_MAX_UNI)) { + sgx_dbg("hwtcl: refused - %u vertex constant(s), the bank " + "carries %u\n", c->vs_const_n, + sgx_hwtcl_limm_uniforms() ? SGX_HWTCL_MAX_UNI_LIMM : + SGX_HWTCL_MAX_UNI); + return 0; + } + /* The buffer the caller set was longer than what is carried, so the + * program would read compiled-in values for the dwords past the end - + * a shading defect rather than a limit. The draw module has the whole + * buffer, so it renders this one correctly. */ + if (c->vs_const_trunc) { + sgx_dbg("hwtcl: refused, the vertex uniforms were truncated to " + "%u dword(s)\n", c->vs_const_n); + return 0; + } + /* A frame is built for one record width, so it is all hardware or all + * draw module - a draw that falls back into a frame the part shaped + * writes eleven-float records where fourteen are read. A caller that + * changes its uniforms between draws therefore turns the hardware + * path off for good, and the draw module takes every draw from here. + * What would lift that is a record per uniform set, and a frame with + * more than one draw record does not survive on this path yet - see + * driver/mesa/. */ + /* Only when nothing can carry the change. The limm scheme gives each + * record its own copy of the vertex program with that record's + * uniforms compiled into it (ctx_upload_vs), so a uniform change + * opens a record rather than ending hardware transform for the life + * of the context - which is what stopped glxgears, whose three gears + * each set their own modelview. */ + if (!sgx_hwtcl_inrec_uniforms() && !sgx_hwtcl_limm_uniforms() && + sgx_vs_constants_changed(&c->ctx)) { + unsigned nd = c->ctx.draws; + + /* One set of uniform values per frame when nothing can carry + * a change, so the frame ends and the next one opens with the + * new values - the part keeps running the transform. Turning + * hardware transform off instead gave it up for the life of + * the context, which is what kept glxgears on the CPU: its + * three gears each set their own modelview. */ + sgx_dbg("hwtcl: the uniforms changed inside a frame; sending " + "%u draw(s) and opening another\n", nd); + sgx_perf_note("uniform values"); + if (sgx_flush(&c->ctx)) + return 0; + if (nd) + sgx_set_clear_enable(&c->ctx, 0); + } + /* One vertex program per frame means one set of uniform values, and a + * frame cannot mix records the part transformed with records the draw + * module did - they are not the same width. So a caller that changes + * its uniforms between draws turns the hardware path off for good and + * the draw module takes every draw from here, including this one. + * Giving each record its own copy of the program is what would lift + * this; see driver/mesa/ for why a copy does not run yet. */ + + for (i = 0; i < c->velems->n; i++) { + const struct pipe_vertex_element *e = &c->velems->el[i]; + + if (!sgx_pipe_attr_comps_fmt((unsigned)e->src_format, + NULL, NULL, NULL)) { + sgx_dbg("hwtcl: refused - element %u is format %u, " + "which the fetch cannot take\n", i, + (unsigned)e->src_format); + return 0; + } + if (e->instance_divisor || e->vertex_buffer_index >= c->nvb) { + sgx_dbg("hwtcl: refused - element %u has divisor %u, " + "buffer %u of %u\n", i, e->instance_divisor, + e->vertex_buffer_index, c->nvb); + return 0; + } + } + /* A shader with no uniforms would need nothing bound, but letting it + * through would put frames with several draw records on the hardware + * path, and those do not come out right yet - see driver/mesa/. */ + if (!c->vs_const_n) + return 0; + return 1; +} + +/* Read one index. */ +static unsigned sgx_hwtcl_index(const void *p, unsigned size, unsigned at) +{ + if (size == 1) + return ((const uint8_t *)p)[at]; + if (size == 2) + return ((const uint16_t *)p)[at]; + return ((const uint32_t *)p)[at]; +} + +/* Build the record the hardware fetches and hand it to the context. Nothing + * here computes: every float is copied from the caller's buffer to the slot + * the compiled program reads it from. */ +/* Returns 0, or -EINVAL for a buffer the fetch cannot be bounded against. + * sgx_pipe_resource_cpu_cached() maps the resource and re-checks its cached + * copy, which is why calling this per corner cost what it did: glmark2's model + * is twenty-one thousand corners, and every element of every one of them came + * through here. */ +static int hwtcl_resolve_elem(struct sgx_pipe_context *c, unsigned e, + struct sgx_hwtcl_elem *o) +{ + const struct pipe_vertex_element *el = &c->velems->el[e]; + const struct pipe_vertex_buffer *vb = &c->vb[el->vertex_buffer_index]; + const char *p = vb->is_user_buffer ? vb->buffer.user : + sgx_pipe_resource_cpu_cached(&c->base, vb->buffer.resource); + + if (!p || !el->src_stride) + return -EINVAL; + if (!vb->is_user_buffer && !vb->buffer.resource) + return -EINVAL; + o->base = p; + o->off = vb->buffer_offset + el->src_offset; + o->stride = el->src_stride; + { + int h = 0, u = 0, bg = 0; + + o->ncomp = sgx_pipe_attr_comps_fmt((unsigned)el->src_format, + &h, &u, &bg); + o->half = (unsigned char)h; + o->u8n = (unsigned char)u; + o->bgra = (unsigned char)bg; + } + /* A resource knows its size; a user buffer does not carry one, and the + * draw module bounds its own fetch, so those go unbounded. */ + o->limit = vb->is_user_buffer ? 0 : vb->buffer.resource->width0; + return 0; +} + +/* Room for a de-duplication of n corners into a table of m slots. */ +static int hwtcl_idx_room(struct sgx_pipe_context *c, unsigned n, unsigned m) +{ + if (n > c->hwidx_n) { + uint16_t *q = realloc(c->hwidx, n * sizeof *q); + + if (!q) + return -1; + c->hwidx = q; + c->hwidx_n = n; + } + if (n > c->hwsrc_n) { + unsigned *q = realloc(c->hwsrc, n * sizeof *q); + + if (!q) + return -1; + c->hwsrc = q; + c->hwsrc_n = n; + } + if (m > c->hwmap_n) { + unsigned *q = realloc(c->hwmap, m * sizeof *q); + + if (!q) + return -1; + c->hwmap = q; + c->hwmap_n = m; + } + return 0; +} + +static int sgx_hwtcl_draw(struct sgx_pipe_context *c, + const struct pipe_draw_info *info, + const struct pipe_draw_start_count_bias *d) +{ + const void *ib = NULL; + unsigned nout = sgx_hwtcl_tri_verts(info->mode, d->count); + unsigned stride_dw, nel; + unsigned ntri = nout / 3, t, k, e; + /* An indexed draw of plain triangles names each vertex about three + * times over, and a record per corner builds each of them that often. + * The part indexes its own records, so build one per vertex of the + * range the draw touches and hand it the caller's indices. Only for + * MESA_PRIM_TRIANGLES, where the corner decomposition is the identity, + * and only smooth-shaded: the decomposition is also what puts the + * provoking vertex where the MTE's flat-shade field expects it, and an + * indexed draw cannot rotate corners. */ + unsigned ncorner = nout, vmin = 0, nvert = 0; + int idxmode = 0; + struct sgx_vertex_buffer sv; + struct sgx_vertex_element se[SGX_MAX_VERTEX_ELEMENTS]; + static int hoist_off = -1; + int no_hoist; + float *rec; + int ret; + + if (hoist_off < 0) + hoist_off = SGX_ENVS("SGX_NO_ELEM_HOIST") != NULL; + no_hoist = hoist_off; + sgx_dbg("hwtcl: mode %u, %u vertices -> %u corners\n", info->mode, + d->count, nout); + if (!nout) + return 0; + /* This path decomposes into triangle corners, and nothing else sets + * the frame's ISP object type - the vbuf owns it, and it is not on + * this path. A frame the draw module left on a sprite object would + * rasterise each corner as a point, which is what a native point + * followed by a transformed triangle would have done. */ + if (c->ctx.prim_objtype != SGX_ISP_OBJ_TRI) { + sgx_perf_note("primitive type"); + if (c->ctx.draws && !sgx_mix_prim() && !sgx_flush(&c->ctx)) + sgx_set_clear_enable(&c->ctx, 0); + c->ctx.prim_objtype = SGX_ISP_OBJ_TRI; + } + if (info->index_size) { + ib = info->has_user_indices ? info->index.user : + sgx_pipe_resource_cpu_cached(&c->base, + info->index.resource); + if (!ib) + return -EINVAL; + } + + /* The width the frame was configured for, not this function's own + * idea of it. sgx_hwtcl_record_dwords() sizes a record at three floats + * a set, which is what a sampled set carries; a set that is four wide + * needs one more, and ctx_hwtcl_record_floats() already widens the + * frame to suit. Building the staging record at the narrower figure + * left the last set off the end of every vertex - a frame with a + * three-float sampled set and a four-float one wants fifteen dwords + * and got fourteen - so the second set's coordinates were never + * written and the draw sampled with whatever the record happened to + * hold. */ + stride_dw = sgx_hwtcl_record_dwords(c->vs_const_n); + if (c->ctx.vtx_floats > stride_dw) + stride_dw = c->ctx.vtx_floats; + if (ib && info->index_size && info->mode == MESA_PRIM_TRIANGLES && + !c->ctx.rast.flatshade && !SGX_ENVS("SGX_NO_HWTCL_INDEX") && + ncorner && ncorner <= c->hwidx_max_n) { + /* De-duplicated, not a min/max range: the caller's indices for + * one draw are scattered across a shared array, so the range + * they span is no smaller than the corner count even where the + * vertices repeat. Measured, a range test took 0 of 52000 + * draws. An open-addressed table over the corners finds the + * repeats themselves, at about twenty cycles a corner against + * the record build it saves. */ + unsigned mask = 1u, q; + + while (mask < ncorner * 2u) + mask <<= 1; + mask -= 1u; + if (hwtcl_idx_room(c, ncorner, mask + 1u)) + return -ENOMEM; + memset(c->hwmap, 0xff, (mask + 1u) * sizeof *c->hwmap); + for (t = 0; t < ncorner; t++) { + unsigned v = sgx_hwtcl_index(ib, info->index_size, + d->start + t) + + (unsigned)d->index_bias; + unsigned h = (v * 2654435761u) & mask; + + while (c->hwmap[h] != 0xffffffffu && + c->hwsrc[c->hwmap[h]] != v) + h = (h + 1u) & mask; + if (c->hwmap[h] == 0xffffffffu) { + c->hwmap[h] = nvert; + c->hwsrc[nvert++] = v; + } + c->hwidx[t] = (uint16_t)c->hwmap[h]; + } + (void)q; + /* Only when it is actually fewer records. */ + if (nvert < ncorner && nvert <= 0xffffu) { + idxmode = 1; + nout = nvert; + } + if (SGX_ENVS("SGX_IDX_STATS")) { + static unsigned n, took, corners, verts; + + n++; took += idxmode; + corners += ncorner; verts += idxmode ? nvert : ncorner; + if (!(n % 2000)) + fprintf(sgx_log(), "sgx: idx: %u draw(s), %u " + "indexed, %u corners -> %u record(s)\n", + n, took, corners, verts); + } + } + { + size_t want = (size_t)nout * stride_dw; + + if (want > c->hwrec_floats) { + float *n = realloc(c->hwrec, want * sizeof *n); + + if (!n) + return -ENOMEM; + c->hwrec = n; + c->hwrec_floats = want; + } + if (c->velems->n > c->hwel_n) { + struct sgx_hwtcl_elem *q = realloc(c->hwel, + c->velems->n * sizeof *q); + + if (!q) + return -ENOMEM; + c->hwel = q; + c->hwel_n = c->velems->n; + } + if (!no_hoist) + for (e = 0; e < c->velems->n; e++) + if (hwtcl_resolve_elem(c, e, &c->hwel[e])) + return -EINVAL; + rec = c->hwrec; + /* Still cleared: the loop below writes the attributes, the + * viewport and the uniforms, and whatever the record's width + * leaves between them has to read as zero. */ + memset(rec, 0, want * sizeof *rec); + } + + for (t = 0; t < (idxmode ? nout : ntri); t++) + for (k = 0; k < (idxmode ? 1u : 3u); k++) { + unsigned i = idxmode ? t : t * 3 + k; + unsigned src; + float *o = rec + (size_t)i * stride_dw; + + if (idxmode) { + /* One record per distinct vertex; the index + * array names them. */ + src = c->hwsrc[i]; + } else { + unsigned at = sgx_hwtcl_corner(info->mode, t, k, + c->ctx.rast.flatshade_first, + c->ctx.rast.flatshade); + + src = d->start + at; + if (info->index_size) { + src = sgx_hwtcl_index(ib, + info->index_size, + d->start + at); + src += (unsigned)d->index_bias; + } + } + /* This draw's uniforms, one copy per vertex, where + * the uniform DMA would otherwise have put them. + * That is what lets consecutive draws with different + * values share a single draw record. */ + if (sgx_hwtcl_inrec_uniforms()) { + unsigned u, un = c->vs_const_n; + + if (un > SGX_HWTCL_MAX_UNI) + un = SGX_HWTCL_MAX_UNI; + for (u = 0; u < un; u++) + o[SGX_HWTCL_MAT_AO + u] = + c->vs_const[u]; + } + + /* The viewport, in the dwords the attributes leave + * free, where the epilogue reads it from. */ + if (SGX_ENVS("SGX_DEBUG") && src == 0) + fprintf(sgx_log(), "sgx: hwtcl viewport: scale " + "%g %g %g translate %g %g %g\n", + c->viewport.scale[0], + c->viewport.scale[1], + c->viewport.scale[2], + c->viewport.translate[0], + c->viewport.translate[1], + c->viewport.translate[2]); + /* A record too narrow to hold it belongs to the + * built-in transform, which reads the viewport out + * of its matrix instead. */ + if (stride_dw < SGX_HWTCL_VP_AO + 6) + goto attribs; + /* With the MTE applying the viewport the program has + * no epilogue to read this, and dwords 8..13 are a + * third attribute's - writing it there would be + * overwritten at best and corrupt the attribute at + * worst. */ + if (sgx_mte_viewport()) + goto attribs; + o[SGX_HWTCL_VP_AO + 0] = c->viewport.scale[0]; + o[SGX_HWTCL_VP_AO + 1] = c->viewport.translate[0]; + o[SGX_HWTCL_VP_AO + 2] = c->viewport.scale[1]; + o[SGX_HWTCL_VP_AO + 3] = c->viewport.translate[1]; + o[SGX_HWTCL_VP_AO + 4] = c->viewport.scale[2]; + o[SGX_HWTCL_VP_AO + 5] = c->viewport.translate[2]; +attribs: + for (e = 0; e < c->velems->n; e++) { + const struct sgx_hwtcl_elem *le = &c->hwel[e]; + struct sgx_hwtcl_elem per_corner; + size_t at; + + /* The old shape, kept so one binary can be + * measured against itself. */ + if (no_hoist) { + if (hwtcl_resolve_elem(c, e, + &per_corner)) + return -EINVAL; + le = &per_corner; + } + at = le->off + (size_t)src * le->stride; + /* The index comes from the caller's buffer, so + * it names any vertex it likes; without this + * the fetch reads outside the vertex buffer + * and the record it builds locks the part. */ + if (le->limit && + at + le->ncomp * (le->half ? 2u : + le->u8n ? 1u : 4u) > + le->limit) + return -EINVAL; + /* Four floats at most, and the count is not a + * constant, so memcpy() here expanded to a + * byte-at-a-time copy - the single hottest + * instruction in a glmark2 frame. Copying as + * floats is what the part wants anyway, and + * an unaligned float load costs nothing on + * this architecture. */ + { + const float *sv = (const float *) + (le->base + at); + const uint16_t *hv = (const uint16_t *) + (le->base + at); + const unsigned char *bv = + (const unsigned char *) + (le->base + at); + float *dv = o + 4 * e; + unsigned q; + + if (le->u8n) { + /* The record is floats + * whatever the caller sent, so + * the byte colour is widened + * here rather than described + * to the fetch. */ + for (q = 0; q < le->ncomp; q++) + dv[q] = bv[q] * + (1.0f / 255.0f); + if (le->bgra) { + float t = dv[0]; + + dv[0] = dv[2]; + dv[2] = t; + } + } else { + for (q = 0; q < le->ncomp; q++) + dv[q] = le->half ? + sgx_half_to_float(hv[q]) + : sv[q]; + } + } + /* A three-component position still needs its + * w, and the program reads all four. */ + if (le->ncomp < 4) + o[4 * e + 3] = 1.0f; + } + } + + if (SGX_ENVS("SGX_DUMP_HWREC")) { + unsigned q; + + fprintf(sgx_log(), "sgx: hwrec stride %u:", stride_dw); + for (q = 0; q < stride_dw && q < 20; q++) + fprintf(sgx_log(), " %g", rec[q]); + fprintf(sgx_log(), "\n"); + } + memset(&sv, 0, sizeof sv); + sv.data = rec; + sv.stride = stride_dw * sizeof *rec; + sv.size = (uint64_t)nout * sv.stride; + /* The record is already in hardware layout, so every dword of it is + * one element and the whole width has to be described - the + * attributes, the viewport and, when they ride here, the uniforms. + * Describing only the first fourteen left everything above them at + * the defaults: the staging record was right and what reached the + * vertex buffer was not. */ + memset(se, 0, sizeof se); + nel = (stride_dw + 3u) / 4u; + if (nel > SGX_MAX_VERTEX_ELEMENTS) + return -ENOSPC; + for (e = 0; e < nel; e++) { + unsigned left = stride_dw - e * 4u; + + se[e].src_offset = e * 4 * sizeof(float); + se[e].ncomp = left > 4u ? 4u : left; + se[e].stride = sv.stride; + se[e].slot = e * 4; + } + /* SGX_HWTCL_FIXED swaps the caller's compiled program for the built-in + * transform, whose uniforms are a row-major matrix that has to end in + * window coordinates. The caller's constants are the columns of its + * own matrix, so this transposes them and folds the viewport in - the + * same arrangement sgxtri renders with. */ + if (SGX_ENVS("SGX_HWTCL_FIXED") && c->vs_const_n >= 16) { + float r[16]; + unsigned i, j; + + for (j = 0; j < 4; j++) + r[12 + j] = c->vs_const[4 * j + 3]; + for (i = 0; i < 3; i++) + for (j = 0; j < 4; j++) + r[4 * i + j] = + c->viewport.scale[i] * + c->vs_const[4 * j + i] + + c->viewport.translate[i] * + c->vs_const[4 * j + 3]; + sgx_set_vs_fixed(&c->ctx, 1); + sgx_set_vs_constants(&c->ctx, r, 16); + } + /* A draw whose uniforms differ from the open record's opens its own: + * a record names the copy of the program those values were compiled + * into. */ + + + /* The record is in hardware layout and holds independent triangle + * corners whatever the caller's mode was, so it can be handed over a + * piece at a time: each piece is a slice of the same staging buffer + * and every piece is whole triangles. + * + * A frame holds a bounded number of triangles and a bounded number of + * vertices, and a draw past either used to be refused as a whole - + * the retry offered the same call again and it was refused again, so + * the geometry was simply lost. glmark2's refract scene draws 69666 + * triangles in one call against a heap sized for 25600 and rendered + * nothing for it. + * + * Only sgx_draw() is repeated, not the record: building it again + * would re-run the viewport and uniform setup around it, which is not + * idempotent. */ + if (idxmode) { + /* One block, indices and all: the chunking below splits by + * record, and an indexed chunk would still name vertices + * outside itself. The whole draw was checked to fit before + * this mode was taken. */ + ret = sgx_set_vertex_buffers(&c->ctx, &sv, 1); + if (!ret) + ret = sgx_set_vertex_elements(&c->ctx, se, nel); + if (!ret) { + c->ctx.draw_idx = c->hwidx; + c->ctx.draw_nidx = ncorner; + ret = sgx_draw(&c->ctx, nout); + if (ret == -ENOSPC && c->ctx.draws) { + sgx_perf_note("hwtcl frame vertex budget"); + if (!sgx_flush(&c->ctx)) { + sgx_set_clear_enable(&c->ctx, 0); + c->ctx.draw_idx = c->hwidx; + c->ctx.draw_nidx = ncorner; + ret = sgx_draw(&c->ctx, nout); + } + } + c->ctx.draw_idx = NULL; + c->ctx.draw_nidx = 0; + } + return ret; + } + { + unsigned done = 0; + + ret = 0; + while (done < nout) { + unsigned room = sgx_draw_room(&c->ctx); + unsigned len = nout - done; + + if (room < 3u || (room < len && !room)) { + /* Nothing fits: the frame goes out and this + * draw opens the one that follows. An empty + * frame that still has no room cannot be made + * to fit and the draw is refused. */ + if (!c->ctx.draws) { + ret = -ENOSPC; + break; + } + sgx_perf_note("hwtcl frame vertex budget"); + if (sgx_flush(&c->ctx)) { + ret = -ENOSPC; + break; + } + sgx_set_clear_enable(&c->ctx, 0); + continue; + } + if (len > room) + len = room; + len -= len % 3u; + + sv.data = rec + (size_t)done * stride_dw; + sv.size = (uint64_t)len * sv.stride; + ret = sgx_set_vertex_buffers(&c->ctx, &sv, 1); + if (!ret) + ret = sgx_set_vertex_elements(&c->ctx, se, nel); + if (!ret) + ret = sgx_draw(&c->ctx, len); + if (ret == -ENOSPC && c->ctx.draws) { + sgx_perf_note("hwtcl frame vertex budget"); + if (sgx_flush(&c->ctx)) + break; + sgx_set_clear_enable(&c->ctx, 0); + continue; + } + if (ret) + break; + done += len; + if (done < nout) + sgx_dbg("hwtcl: %u of %u corner(s) placed, the " + "rest opens another frame\n", done, + nout); + } + } + if (sgx_debug()) { + unsigned q; + + for (q = 0; q < 3 && q < nout; q++) { + const float *r = rec + (size_t)q * SGX_VTX_FLOATS_HWTCL; + + fprintf(sgx_log(), "sgx: rec %u:", q); + for (e = 0; e < stride_dw; e++) + fprintf(sgx_log(), " %g", r[e]); + fprintf(sgx_log(), "\n"); + } + } + if (sgx_hud_trace() && nout && nout <= 6u) { + unsigned q; + + fprintf(sgx_log(), "sgx: hud: hwtcl draw %u vtx ret %d records %u " + "vtx_floats %u | v0", nout, ret, c->ctx.nrange, + c->ctx.vtx_floats); + for (q = 0; q < stride_dw; q++) + fprintf(sgx_log(), " %g", rec[q]); + fputc('\n', stderr); + } + sgx_dbg("hwtcl: the part ran the vertex program, %u element(s), %d\n", + c->velems->n, ret); + return ret; +} + +/* pipe_context::draw_vbo. Gallium batches several draws into one call, and + * each is a separate primitive range, so each is a separate draw here. */ +/* Hand the draw module the texels of every view the vertex stage samples. + * + * Only level zero: a vertex texture2D has no derivatives to take a level + * from, so that is the one it reads, and mapping the level rather than the + * object gives a linear copy of a twiddled one for free. Returns the number + * mapped, which the caller unmaps again once the draw is through the module. + */ +static unsigned sgx_pipe_map_vtx_textures(struct sgx_pipe_context *c) +{ + unsigned i, n = 0; + + for (i = 0; i < c->nvtx_views; i++) { + struct pipe_sampler_view *v = c->vtx_views[i]; + struct sgx_pipe_resource *r; + uint32_t row[PIPE_MAX_TEXTURE_LEVELS] = { 0 }; + uint32_t img[PIPE_MAX_TEXTURE_LEVELS] = { 0 }; + uint32_t off[PIPE_MAX_TEXTURE_LEVELS] = { 0 }; + uint32_t stride = 0; + void *p = NULL; + + if (!v || !v->texture) + continue; + r = (struct sgx_pipe_resource *)v->texture; + /* Read-only: the draw module samples this on the CPU and + * writes nothing, so the staging copy is kept between draws + * rather than untwiddled and twiddled back on each one. */ + if (sgx_resource_map_level_ro(&r->r, c->ctx.ws, 0, &p, + &stride) != SGX_RESOURCE_OK || !p) + continue; + row[0] = stride; + img[0] = stride * r->r.level[0].height; + sgx_dbg("vtx tex %u: %ux%u stride %u twiddled %u fmt %d\n", i, + r->r.level[0].width, r->r.level[0].height, stride, + r->r.twiddled, (int)r->r.format); + draw_set_mapped_texture(c->draw, MESA_SHADER_VERTEX, i, + r->r.level[0].width, + r->r.level[0].height, 1, 0, 0, 1, 0, + p, row, img, off); + n = i + 1u; + } + return n; +} + +/* And put them back: a twiddled level's mapping is a staging copy. */ +static void sgx_pipe_unmap_vtx_textures(struct sgx_pipe_context *c, + unsigned n) +{ + unsigned i; + + for (i = 0; i < n; i++) + if (c->vtx_views[i] && c->vtx_views[i]->texture) + sgx_resource_unmap_level( + &((struct sgx_pipe_resource *) + c->vtx_views[i]->texture)->r); +} + +static void sgx_pipe_draw_vbo_inner(struct pipe_context *pc, + const struct pipe_draw_info *info, + unsigned drawid_offset, + const struct pipe_draw_indirect_info *indirect, + const struct pipe_draw_start_count_bias *draws, + unsigned num_draws); + +/* Rebuild the descriptor of any unit whose resource has moved since it was + * bound. + * + * A texture keeps one address for its life, except when it is also a render + * target: taking the depth or target slot moves it there and giving the slot + * up moves it back. glmark2's shadow scene binds its shadow map to a unit + * while that map is still the depth attachment, so the descriptor named the + * depth slot - and by the pass that samples it the slot held the window's own + * depth buffer, a different surface entirely. The compare then answered the + * same thing for every pixel and the scene came out black. + * + * The view itself is unchanged, so re-applying it is a rebuild at the address + * the resource has now; nothing here flushes, because the unit's object is + * the same one. */ +static void sgx_pipe_resettle_views(struct sgx_pipe_context *c) +{ + unsigned i; + + for (i = 0; i < c->ncur_views && i < SGX_MAX_TEX_UNITS; i++) { + struct pipe_sampler_view *pv = c->cur_views[i]; + struct sgx_pipe_resource *r; + + if (!pv || !pv->texture) + continue; + r = (struct sgx_pipe_resource *)pv->texture; + if (!r->r.bo.gpu_va || r->r.bo.gpu_va == c->view_base[i]) + continue; + sgx_dbg("view %u: resource moved 0x%llx -> 0x%llx, rebuilt\n", + i, (unsigned long long)c->view_base[i], + (unsigned long long)r->r.bo.gpu_va); + sgx_pipe_set_sampler_views(&c->base, MESA_SHADER_FRAGMENT, i, 1, + 0, &c->cur_views[i]); + } +} + +/* Timed as a whole - the transform the draw module runs for a program the part + * cannot take is inside here, and the report has no other way to see it. */ +static void sgx_pipe_draw_vbo(struct pipe_context *pc, + const struct pipe_draw_info *info, + unsigned drawid_offset, + const struct pipe_draw_indirect_info *indirect, + const struct pipe_draw_start_count_bias *draws, + unsigned num_draws) +{ + uint64_t t = sgx_perf_mark(); + + sgx_pipe_draw_vbo_inner(pc, info, drawid_offset, indirect, draws, + num_draws); + sgx_perf_add(5, t); +} + +static void sgx_pipe_draw_vbo_inner(struct pipe_context *pc, + const struct pipe_draw_info *info, + unsigned drawid_offset, + const struct pipe_draw_indirect_info *indirect, + const struct pipe_draw_start_count_bias *draws, + unsigned num_draws) +{ + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + unsigned i; + + sgx_dbg("draw_vbo: %u draw(s), mode %u, %u vertex buffer(s)\n", + num_draws, info ? info->mode : 0, c->nvb); + sgx_pipe_resettle_views(c); + /* A sample mask that selects no sample writes no fragment. The part + * has no per-sample write mask to program, so it is honoured here - + * see sgx_pipe_set_sample_mask() for what the three cases are and why + * a partial mask lands here too. */ + if (sgx_sample_mask_blocks(c)) { + sgx_dbg("draw_vbo: the sample mask 0x%x selects no sample of " + "%u, so nothing is drawn\n", c->sample_mask, + c->fb_samples ? c->fb_samples : 1u); + return; + } + /* glCullFace(GL_FRONT_AND_BACK) rasterises no polygon - points and + * lines still draw. The part's cull field is three-valued and the + * translation reports both as the back, so this is the one place the + * request can be honoured at all. */ + if (info && c->cur_rast && + ((const struct sgx_pipe_rast *)c->cur_rast)->pipe.cull_face == + PIPE_FACE_FRONT_AND_BACK && + u_reduced_prim(info->mode) == MESA_PRIM_TRIANGLES) { + sgx_dbg("draw_vbo: both faces culled, nothing to rasterise\n"); + return; + } + /* A frame that fell back to the draw module gets the part's transform + * back here, where the frame is empty and changing the record's width + * throws no geometry away. */ + if (c->hwtcl_rearm && !c->hwtcl && !c->ctx.draws) { + c->hwtcl_rearm = 0; + c->hwtcl = 1; + sgx_pipe_settle_hwtcl(c); + /* And bind for it: the shader in hand was bound while the + * flag was down, so it is the pass-through. */ + sgx_pipe_apply_vs(c); + sgx_dbg("draw_vbo: a new frame, so the part's transform is " + "back on\n"); + } + sgx_pipe_settle_borders(c); + /* A shadow sampler's comparison is lowered into the program ahead of + * everything else, from the NIR the program keeps, so a change of + * the bound comparison rebuilds it from there first - the keys + * below are then checked against the fresh build. */ + if (c->cur_fs && ((struct sgx_shader *)c->cur_fs)->src_nir && + sgx_shader_cmp_differs((struct sgx_shader *)c->cur_fs, + c->cmp, c->ncmp)) { + if (c->ctx.draws) { + unsigned nd = c->ctx.draws; + + sgx_perf_note("shadow compare rebuild"); + sgx_flush(&c->ctx); + sgx_set_clear_enable(&c->ctx, 0); + sgx_dbg("draw_vbo: sent %u draw(s) before rebuilding " + "for the shadow comparison\n", nd); + } + if (sgx_pipe_fs_rebuild_cmp(c, (struct sgx_shader *)c->cur_fs)) { + sgx_dbg("draw_vbo: no program for the bound shadow " + "comparison\n"); + return; + } + sgx_bind_fs(&c->ctx, (const struct sgx_shader *)c->cur_fs); + } + /* The unit's REPEAT wraps at the padded power of two, so a repeating + * sample of a texture that is not one has to wrap in the program. + * Which units those are is a property of what is bound, so it is baked + * into the program and rebuilt when it changes, the same way the sRGB + * mask is. Usually nothing is bound that needs it, and then this costs + * the walk over three units and a comparison. */ + if (c->cur_fs) { + unsigned npot = sgx_npot_wrap_mask(c); + + if (((struct sgx_shader *)c->cur_fs)->npot_key != npot && + sgx_shader_retarget_npot((struct sgx_shader *)c->cur_fs, + npot)) { + sgx_dbg("draw_vbo: no program for the wrap of units " + "%x\n", npot); + return; + } + } + /* An sRGB view has to be converted to linear as it is sampled. The + * texture unit does that itself under DOUTT0 GAMMA; with the program + * decoding instead (SGX_SRGB_SHADER) the program depends on which + * units carry one. Usually none, and then this costs a comparison. */ + if (c->cur_fs && + ((struct sgx_shader *)c->cur_fs)->srgb_key != c->srgb_mask && + sgx_shader_retarget_srgb((struct sgx_shader *)c->cur_fs, + c->srgb_mask)) { + sgx_dbg("draw_vbo: no program for sRGB units %x\n", + c->srgb_mask); + return; + } + /* A depth view is sampled as the four bytes the ISP stored and the + * depth put back together in the program, so which units carry one is + * baked in the same way. */ + if (c->cur_fs && + ((struct sgx_shader *)c->cur_fs)->depth_key != c->depth_mask && + sgx_shader_retarget_depth((struct sgx_shader *)c->cur_fs, + c->depth_mask)) { + sgx_dbg("draw_vbo: no program for depth units %x\n", + c->depth_mask); + return; + } + /* And the same for what the views make of the channels they fetched. + * An alpha-only view reads (0,0,0,a), which this part cannot do in the + * texture unit, so the program carries it. + * + * Rebuilding replaces the code the frame's records were built from, so + * the frame ends first - the records already in it name the old + * program, and leaving them to run against the new one stalled the + * render on every frame of the ideas scene. */ + if (c->cur_fs && + sgx_shader_swz_differs((struct sgx_shader *)c->cur_fs, + c->swz, c->nswz)) { + if (c->ctx.draws) { + unsigned nd = c->ctx.draws; + + sgx_perf_note("texture swizzle rebuild"); + sgx_flush(&c->ctx); + sgx_set_clear_enable(&c->ctx, 0); + sgx_dbg("draw_vbo: sent %u draw(s) before rebuilding " + "for the texture swizzles\n", nd); + } + if (sgx_shader_retarget_swz((struct sgx_shader *)c->cur_fs, + c->swz, c->nswz)) { + sgx_dbg("draw_vbo: no program for the bound texture " + "swizzles\n"); + return; + } + /* The context holds what the old form declared - register + * counts and the secondary layout among it - so the rebuilt + * program has to be installed again. */ + sgx_bind_fs(&c->ctx, (const struct sgx_shader *)c->cur_fs); + } + /* And for the winding a program reading gl_FrontFacing was built + * for: glFrontFace changes it, and the answer is compiled in. */ + if (c->cur_fs && c->cur_rast && + ((struct sgx_shader *)c->cur_fs)->face_in >= 0) { + struct sgx_shader *fs = c->cur_fs; + unsigned want = sgx_face_swap_of( + &((const struct sgx_pipe_rast *)c->cur_rast)->hw); + + if (fs->face_key != want) { + if (c->ctx.draws) { + unsigned nd = c->ctx.draws; + + sgx_perf_note("front face rebuild"); + sgx_flush(&c->ctx); + sgx_set_clear_enable(&c->ctx, 0); + sgx_dbg("draw_vbo: sent %u draw(s) before " + "rebuilding for the front face\n", nd); + } + /* Only when there is no program left to draw with: + * a failed compile clears the object in place. A + * rebuild that merely came out over budget keeps a + * valid program and returns 0, because dropping the + * one draw that triggered it - and no other - loses + * a quad rather than enforcing anything. */ + if (sgx_shader_retarget_face(fs, want)) { + fprintf(sgx_log(), "sgx: the front face " + "rebuild left no program; this draw " + "cannot be made\n"); + return; + } + sgx_bind_fs(&c->ctx, fs); + } + } + /* And for what the views' fetches return: a float texel is unpacked + * differently and a texel in planes takes a state block a plane, so + * the sampler layout changes with it - the frame ends first for the + * same reason as above. */ + if (c->cur_fs && + sgx_shader_cls_differs((struct sgx_shader *)c->cur_fs, + c->cls, c->ncls)) { + if (c->ctx.draws) { + unsigned nd = c->ctx.draws; + + sgx_perf_note("texel class rebuild"); + sgx_flush(&c->ctx); + sgx_set_clear_enable(&c->ctx, 0); + sgx_dbg("draw_vbo: sent %u draw(s) before rebuilding " + "for the texel formats\n", nd); + } + if (sgx_shader_retarget_cls((struct sgx_shader *)c->cur_fs, + c->cls, c->ncls)) { + sgx_dbg("draw_vbo: no program for the bound texel " + "formats\n"); + return; + } + sgx_bind_fs(&c->ctx, (const struct sgx_shader *)c->cur_fs); + } + /* The vertex data has to be where the draw module can read it, which + * is the CPU when it is the one running the vertex shader. */ + for (i = 0; i < c->nvb; i++) { + const struct pipe_vertex_buffer *vb = &c->vb[i]; + const void *p = NULL; + + if (vb->is_user_buffer) + p = vb->buffer.user; + else if (vb->buffer.resource) + p = sgx_pipe_resource_cpu(pc, vb->buffer.resource); + /* The size the draw module clamps its reads to. A resource + * knows its own extent, and saying it is unbounded let an + * index past the end of the buffer read off it. */ + draw_set_mapped_vertex_buffer(c->draw, i, p, + !p ? 0 : + (!vb->is_user_buffer && vb->buffer.resource) ? + (size_t)vb->buffer.resource->width0 : ~(size_t)0); + } + if (info->index_size) { + const void *p = info->has_user_indices ? info->index.user : + sgx_pipe_resource_cpu(pc, info->index.resource); + + /* Likewise the index buffer: this is the only bound the + * draw module applies to an index read, and a count past + * the end of the buffer walked off it. A user pointer + * carries no size, so only a resource can be bounded. */ + /* A resource that would not map leaves this null, and the + * module then fetches every index through it. There is no + * meaningful draw to make without the indices. */ + if (!p) { + sgx_dbg("draw: the index buffer would not map; the " + "draw is dropped\n"); + return; + } + draw_set_indexes(c->draw, p, info->index_size, + (!info->has_user_indices && + info->index.resource) ? + info->index.resource->width0 : ~0u); + } + + /* A frame holds one fragment program, one texture and one blend, and a + * draw needing a different one of them cannot join it. Ending the + * frame here would be the answer if a frame could be made to preserve + * what the last one left, and it cannot yet - see the note on + * sgx_set_clear_enable(). Until then the draws merge and the last + * state wins, which is reported rather than hidden. */ + /* The fragment uniforms are in that list too. They live in one heap + * block the secondary PDS reads, so a draw that changes them and joins + * the record before it makes both draws render with the last values - + * two quads at different depths came out the same colour, which reads + * exactly like the depth test having done nothing. */ + if (c->ctx.draws && + (c->cur_fs != c->frame_fs || c->cur_blend != c->frame_blend || + c->cur_tex != c->frame_tex || + c->fs_const_serial != c->frame_fs_const_serial || + c->cur_dsa != c->frame_dsa || c->cur_rast != c->frame_rast || + memcmp(&c->stencil_ref, &c->frame_sref, sizeof c->frame_sref) || + c->blend_const_serial != c->frame_blend_const_serial || + /* The shape too, now that a record carries its own: draws of + * different shapes share their state but not their block. */ + (c->ctx.nrange && + c->ctx.range[c->ctx.nrange - 1].objtype != c->ctx.prim_objtype))) { + sgx_dbg("draw: the state changed, so this draw gets its own " + "record\n"); + sgx_next_draw_record(&c->ctx); + } + c->frame_fs = c->cur_fs; + c->frame_fs_const_serial = c->fs_const_serial; + c->frame_blend_const_serial = c->blend_const_serial; + c->frame_blend = c->cur_blend; + c->frame_tex = c->cur_tex; + c->frame_dsa = c->cur_dsa; + c->frame_rast = c->cur_rast; + c->frame_sref = c->stencil_ref; + + /* An indirect draw takes its count from a buffer the hardware would + * have to read. Dropping it is better than drawing something else. */ + if (indirect) { + sgx_dbg("draw_vbo: an indirect draw has no path here\n"); + return; + } + sgx_dbg("draw_vbo: mode %u, %u range(s), %u instance(s), %u buffer(s)\n", + info->mode, num_draws, info->instance_count, c->nvb); + if (SGX_ENVS("SGX_DEBUG")) { + const uint32_t *h = c->ctx.bo[SGX_CTX_BO_HEAP].map; + + if (h) + sgx_dbg("draw_vbo: fs uni serial %u: %08x %08x %08x %08x\n", + c->fs_const_serial, + h[SGX_FS_CONST_OFF / 4 + 0], + h[SGX_FS_CONST_OFF / 4 + 1], + h[SGX_FS_CONST_OFF / 4 + 2], + h[SGX_FS_CONST_OFF / 4 + 3]); + } + + /* SGX_SPLIT_EVERY starts a new draw record every N draws, with nothing + * else changing between them, so that the record count is the only + * variable. It is the reproducer for the multi-record failure recorded + * in driver/mesa/README.md, on either transform path. */ + /* The counter restarts with each frame, so the split lands on the same + * draw every time and the position is a variable rather than something + * that walks. */ + if (c->split_every && ++c->draws_seen == c->split_every) + sgx_next_draw_record(&c->ctx); + + /* The part runs the vertex program when it can. What it cannot do - + * a shader whose uniforms outgrow the primary attribute bank, a + * primitive the draw record cannot express - goes back through the + * draw module, and says so rather than silently costing frames. */ + if (sgx_hwtcl_can(c, info)) { + int ret = 0; + + for (i = 0; i < num_draws && !ret; i++) + ret = sgx_hwtcl_draw(c, info, &draws[i]); + if (!ret) + return; + /* The draw module's path flushes and retries here; this one + * has never had a continuation, so the geometry is gone. */ + if (sgx_hud_trace()) + fprintf(sgx_log(), "sgx: hud: the hardware path dropped a " + "draw of %u vertices: %d\n", + draws[0].count, ret); + sgx_dbg("draw_vbo: the hardware path returned %d\n", ret); + return; + } + /* A frame is built for one vertex record width, so it is all hardware + * or all draw module: a draw that falls back into a frame the part + * shaped writes eleven-float records where fourteen are read, and what + * comes out is neither path's picture. So the first draw that cannot + * go to the part takes the rest of the frame back to the draw module - + * after the frame it already holds has gone, because changing the + * width rebuilds the frame and would throw that geometry away. */ + /* Only when the frame really is in hardware shape. A program pair the + * part cannot run has already put the context back to the narrow + * record at bind time, so there is nothing to flush and nothing to + * turn off - and flushing anyway split a frame per draw. */ + if (c->hwtcl && c->ctx.hwtcl) { + unsigned nd = c->ctx.draws; + + sgx_perf_note("transform fallback"); + sgx_flush(&c->ctx); + /* Only when something went: a flush with nothing in it leaves + * the caller's pending clear pending, and turning clearing off + * then loses it. */ + if (nd) + sgx_set_clear_enable(&c->ctx, 0); + sgx_dbg("draw_vbo: mode %u cannot go to the part, so the draw " + "module takes this frame\n", info->mode); + c->hwtcl = 0; + sgx_set_hwtcl(&c->ctx, 0); + /* For this frame, not for the context: the width is what a + * frame cannot change, and the next frame is a new one. A + * scene that mixes draws the part can take with draws it + * cannot - ioquake3's, whose lightmapped surfaces have a + * fourth attribute - used to lose the transform for the life + * of the process on the first of them, whichever way round + * they came. It now costs one extra frame boundary, once per + * frame, because once the frame is on the draw module every + * later draw in it joins there. SGX_HWTCL_REARM=0 keeps the + * old behaviour. */ + c->hwtcl_rearm = c->hwtcl_want && + !sgx_pipe_env_off("SGX_HWTCL_REARM"); + } + { + /* The stipple stage binds its pattern texture and sampler on + * a unit of its own during the draw and, at its flush, + * restores the caller's sampler state by the count the state + * tracker last set (draw_pipe_pstipple.c pstip_flush) - which + * never reaches the unit it used, so the pattern stayed bound + * there for every draw after, with a program that samples + * nothing. What was on that unit before the draw goes back. */ + const struct sgx_pipe_rast *sr = c->cur_rast; + int stip = sr && sr->pipe.poly_stipple_enable; + struct pipe_sampler_view *had_view[SGX_MAX_TEX_UNITS]; + void *had_sampler[SGX_MAX_TEX_UNITS]; + + if (stip) { + memcpy(had_view, c->cur_views, sizeof had_view); + memcpy(had_sampler, c->cur_samplers, sizeof had_sampler); + } + { + unsigned nvt = sgx_pipe_map_vtx_textures(c); + + draw_vbo(c->draw, info, drawid_offset, indirect, draws, + num_draws, 0); + draw_flush(c->draw); + sgx_pipe_unmap_vtx_textures(c, nvt); + } + if (stip && c->stipple_unit >= 0 && + c->stipple_unit < (int)SGX_MAX_TEX_UNITS && + (c->cur_views[c->stipple_unit] != + had_view[c->stipple_unit] || + c->cur_samplers[c->stipple_unit] != + had_sampler[c->stipple_unit])) { + unsigned u = (unsigned)c->stipple_unit; + + sgx_dbg("draw_vbo: putting unit %u back after the " + "stipple stage\n", u); + sgx_pipe_bind_sampler_states(pc, MESA_SHADER_FRAGMENT, + u, 1, &had_sampler[u]); + sgx_pipe_set_sampler_views(pc, MESA_SHADER_FRAGMENT, + u, 1, 0, &had_view[u]); + } + } + /* SGX_ONE_DRAW ends the frame after every draw, so a defect that only + * shows when several records share a frame can be told from one in the + * record itself. */ + if (SGX_ENVS("SGX_ONE_DRAW")) { + sgx_perf_note("one draw (debug)"); + sgx_flush(&c->ctx); + } +} + +/* pipe_context::buffer_map and ::texture_map. */ +static void *sgx_pipe_buffer_map(struct pipe_context *pc, + struct pipe_resource *res, unsigned level, + unsigned usage, const struct pipe_box *box, + struct pipe_transfer **out) +{ + sgx_xfer_bufmap++; + struct sgx_pipe_context *c = (struct sgx_pipe_context *)pc; + /* The resource is the wrapper, not the driver object inside it. This + * used to cast pipe_resource straight to sgx_resource, which reads the + * wrapper's own fields as a bo - harmless only for as long as nothing + * called it, and glMapBufferRange calls it. */ + struct sgx_pipe_resource *r = (struct sgx_pipe_resource *)res; + struct pipe_transfer *t; + void *p = NULL; + + (void)level; + (void)usage; + if (out) + *out = NULL; + if (!r || !out || !box) + return NULL; + if (SGX_ENVS("SGX_MAP_STATS")) { + static unsigned n, uns, disc, wr; + + n++; + uns += (usage & PIPE_MAP_UNSYNCHRONIZED) != 0; + disc += (usage & (PIPE_MAP_DISCARD_RANGE | + PIPE_MAP_DISCARD_WHOLE_RESOURCE)) != 0; + wr += (usage & PIPE_MAP_WRITE) != 0; + if (!(n % 500)) + fprintf(sgx_log(), "sgx: buffer_map %u: unsync %u " + "discard %u write %u\n", n, uns, disc, wr); + } + /* An unsynchronized map does not wait: the caller has said the part is + * not reading what it is about to write, and Mesa's stream uploader + * says it for every client array it uploads. */ + if (sgx_resource_map_nowait(&r->r, c->ctx.ws, &p, + (usage & PIPE_MAP_UNSYNCHRONIZED) != 0) != + SGX_RESOURCE_OK) + return NULL; + /* A caller that gets a pointer and no transfer has nothing to unmap + * with, so the mapping would be leaked and the unmap would fault. */ + /* A threaded_transfer even here: the threaded context frees every + * transfer through its own slab and reads the fields past + * pipe_transfer, so a bare one is too small. */ + { + struct threaded_transfer *tt = CALLOC_STRUCT(threaded_transfer); + + t = tt ? &tt->b : NULL; + } + if (!t) { + sgx_resource_unmap(&r->r); + return NULL; + } + t->resource = NULL; + pipe_resource_reference(&t->resource, res); + t->level = level; + t->usage = usage; + t->box = *box; + t->stride = 0; + t->layer_stride = 0; + *out = t; + if (usage & PIPE_MAP_WRITE) { + r->write_mapped++; + sgx_pipe_shadow_dirty(res); + } + return (char *)p + box->x; +} + +static void sgx_pipe_buffer_unmap(struct pipe_context *pc, + struct pipe_transfer *t) +{ + struct sgx_pipe_resource *r; + + (void)pc; + if (!t) + return; + r = (struct sgx_pipe_resource *)t->resource; + if (r) { + /* Whether or not the map said write: a read-only one costs a + * bulk copy to re-take and a missed write costs a wrong + * frame. */ + if ((t->usage & PIPE_MAP_WRITE) && r->write_mapped) + r->write_mapped--; + sgx_pipe_shadow_dirty(t->resource); + sgx_resource_unmap(&r->r); + } + pipe_resource_reference(&t->resource, NULL); + FREE(t); +} + +struct pipe_context *sgx_context_create(struct pipe_screen *screen, + void *priv, unsigned flags); + +struct pipe_context *sgx_context_create(struct pipe_screen *screen, + void *priv, unsigned flags) +{ + struct sgx_pipe_screen *s = (struct sgx_pipe_screen *)screen; + struct sgx_pipe_context *c; + + (void)priv; + (void)flags; + /* No winsys means no device. Handing back a context that cannot submit + * would fail at the first flush instead of here. */ + if (!s || !s->ws) + return NULL; + c = calloc(1, sizeof(*c)); + if (!c) + return NULL; + c->screen = s; + c->next = s->ctxs; + s->ctxs = c; + c->base.screen = screen; + c->base.priv = priv; + sgx_context_init(&c->ctx, s->ws); + + /* Hardware transform, on by default. This only says the driver may + * try; sgx_pipe_settle_hwtcl() decides per program pair whether the + * part can actually run the transform, and anything it cannot goes + * back through the draw module. SGX_HWTCL=0 turns it off outright. */ + { + const char *e; + + c->hwtcl = c->hwtcl_want = sgx_hwtcl_wanted(); + e = SGX_ENVS("SGX_SPLIT_EVERY"); + c->split_every = (e && *e) ? (unsigned)atoi(e) : 0; + } + c->stipple_unit = -1; + c->draw = draw_create(&c->base); + if (!c->draw) { + free(c); + return NULL; + } + /* The part rasterises triangles only, and a point or a line reaching + * the backend had no path and was dropped - which is every white thing + * glamor draws for PolyPoint, PolyLine, PolySegment, PolyRectangle and + * the text cursor. A zero threshold has the draw module expand them + * into triangles instead. */ + /* The draw module's own wide-point stage sizes its quad against the + * viewport, and by the time a vertex reaches it this driver's epilogue + * has already put the position in window coordinates - so a one-pixel + * point came out twenty-three pixels wide. The X server draws every + * core-font glyph as one point per lit bit of the glyph bitmap, so + * that turned each letter into a filled block. Points are widened here + * instead, in the space they are already in. */ + draw_wide_point_threshold(c->draw, 1e30f); + /* And lines are widened here too, or by the ISP itself. A zero + * threshold had the draw module turn every line wider than one pixel + * into triangles before the backend saw a line at all, so + * EURASIA_ISPA_PLWIDTH - the one field the part has for a line's + * width, which this driver already computes - was never exercised at + * any setting. The field holds sixteen pixels and the line-width cap + * says so, so nothing should ask for more; whatever does is left to + * the module. sgx_vbuf_widen_lines() widens the rest on the CPU when + * the ISP is not rasterising the line itself, at the same width and + * with the band snapped to whole pixels. */ + /* Handing wide lines to the ISP costs a core recovery: measured, the + * linewidth and lineband cases cost three each, and three runs of a + * suite sweep aborted the client outright - while every line case + * still reported a pass. A wrong answer that looks right, which is why + * the recoveries and not the pixels decided this. Off until the cause + * is found; SGX_ISP_LINES=1 hands them to the ISP. */ + draw_wide_line_threshold(c->draw, + SGX_ENVS("SGX_ISP_LINES") ? + (float)SGX_ISP_PLWIDTH_MAX : 0.0f); + /* And not for sprites either: the coordinates are written by the same + * widening, see sgx_pipe_create_rast(). */ + draw_enable_point_sprites(c->draw, false); + c->vbuf = sgx_vbuf_create(&c->ctx, c->draw); + if (!c->vbuf) { + draw_destroy(c->draw); + free(c); + return NULL; + } + { + struct draw_stage *stage = draw_vbuf_stage(c->draw, c->vbuf); + + if (!stage) { + c->vbuf->destroy(c->vbuf); + draw_destroy(c->draw); + free(c); + return NULL; + } + draw_set_rasterize_stage(c->draw, stage); + /* The stage and the backend are set separately; the fast path + * that bypasses the pipeline goes straight to the backend. */ + draw_set_render(c->draw, c->vbuf); + } + + /* Not u_upload_create_default(): its megabyte is the whole raster + * geometry window, which also holds the frame's own vertex and index + * block, so the very first upload a GL client needed was refused and + * nothing could draw. A quarter of that leaves room for several live + * at once. */ + c->base.stream_uploader = + u_upload_create(&c->base, 256 * 1024, + PIPE_BIND_VERTEX_BUFFER | + PIPE_BIND_INDEX_BUFFER | + PIPE_BIND_CONSTANT_BUFFER, + PIPE_USAGE_STREAM, 0); + c->base.const_uploader = c->base.stream_uploader; + + c->base.create_depth_stencil_alpha_state = sgx_pipe_create_dsa; + c->base.bind_depth_stencil_alpha_state = sgx_pipe_bind_dsa; + c->base.delete_depth_stencil_alpha_state = sgx_pipe_delete_state; + c->base.create_rasterizer_state = sgx_pipe_create_rast; + c->base.bind_rasterizer_state = sgx_pipe_bind_rast; + c->base.delete_rasterizer_state = sgx_pipe_delete_state; + c->base.create_fs_state = sgx_pipe_create_fs; + c->base.bind_fs_state = sgx_pipe_bind_fs; + c->base.delete_fs_state = sgx_pipe_delete_fs; + c->base.create_vs_state = sgx_pipe_create_vs; + c->base.bind_vs_state = sgx_pipe_bind_vs; + c->base.delete_vs_state = sgx_pipe_delete_vs; + c->base.draw_vbo = sgx_pipe_draw_vbo; + c->base.create_vertex_elements_state = sgx_pipe_create_velems; + c->base.bind_vertex_elements_state = sgx_pipe_bind_velems; + c->base.delete_vertex_elements_state = sgx_pipe_delete_velems; + c->base.set_vertex_buffers = sgx_pipe_set_vertex_buffers; + c->base.set_framebuffer_state = sgx_pipe_set_framebuffer; + c->base.buffer_map = sgx_pipe_buffer_map; + c->base.buffer_unmap = sgx_pipe_buffer_unmap; + c->base.flush = sgx_pipe_flush; + c->base.get_device_reset_status = sgx_pipe_get_device_reset_status; + /* Level with the winsys: a context created after another one lost a + * frame has not lost anything itself, and starting at zero would have + * it report that earlier loss as its own. */ + c->reset_gen = sgx_winsys_lost_gen(c->ctx.ws); + /* Every sample selected, which is where GL starts and what a state + * tracker that never mentions the mask leaves in force. Zero here + * would have drawn nothing until the first set_sample_mask arrived. */ + c->sample_mask = ~0u; + /* The most corners a draw may have and still be de-duplicated into + * per-vertex records: comfortably more than this driver sees - + * ioquake3's draws are a couple of hundred - and small enough that the + * record block always fits without chunking. */ + c->hwidx_max_n = 4096; + c->fb_samples = 1; + sgx_dbg("context create\n"); + c->base.destroy = sgx_pipe_context_destroy; + c->base.clear = sgx_pipe_clear; + c->base.set_viewport_states = sgx_pipe_set_viewport_states; + c->base.set_scissor_states = sgx_pipe_set_scissor_states; + c->base.create_blend_state = sgx_pipe_create_blend; + c->base.bind_blend_state = sgx_pipe_bind_blend; + c->base.delete_blend_state = sgx_pipe_delete_state; + c->base.set_blend_color = sgx_pipe_set_blend_color; + c->base.set_stencil_ref = sgx_pipe_set_stencil_ref; + c->base.set_sample_mask = sgx_pipe_set_sample_mask; + c->base.set_constant_buffer = sgx_pipe_set_constant_buffer; + c->base.generate_mipmap = sgx_pipe_generate_mipmap; + c->base.set_sampler_views = sgx_pipe_set_sampler_views; + c->base.create_sampler_view = sgx_pipe_create_sampler_view; + c->base.sampler_view_destroy = sgx_pipe_sampler_view_destroy; + /* Not optional, whatever its absence from most drivers suggests: + * st_destroy_context() calls it without checking, so a context that + * is torn down - which is every context glamor opens, since it makes + * a desktop GL one first and drops it - jumps through a null pointer. */ + c->base.sampler_view_release = u_default_sampler_view_release; + /* Mesa calls this through the context with no NULL check + * (pipe_resource_release(), u_inlines.h), so leaving it unset is a + * jump to zero the moment a client deletes a buffer - glDeleteBuffers + * segfaulted every time. The default just drops the reference. */ + c->base.resource_release = u_default_resource_release; + c->base.create_sampler_state = sgx_pipe_create_sampler_state; + c->base.bind_sampler_states = sgx_pipe_bind_sampler_states; + c->base.delete_sampler_state = sgx_pipe_delete_state; + c->base.texture_map = sgx_pipe_texture_map; + c->base.texture_unmap = sgx_pipe_texture_unmap; + c->base.buffer_subdata = sgx_pipe_buffer_subdata; + c->base.texture_subdata = sgx_pipe_texture_subdata; + c->base.texture_barrier = sgx_pipe_texture_barrier; + c->base.clear_render_target = sgx_pipe_clear_render_target; + c->base.memory_barrier = sgx_pipe_memory_barrier; + c->base.set_polygon_stipple = sgx_pipe_set_polygon_stipple; + c->base.set_clip_state = sgx_pipe_set_clip_state; + c->base.set_min_samples = sgx_pipe_set_min_samples; + c->base.set_active_query_state = sgx_pipe_set_active_query_state; + c->base.create_query = sgx_pipe_create_query; + c->base.destroy_query = sgx_pipe_destroy_query; + c->base.begin_query = sgx_pipe_begin_query; + c->base.end_query = sgx_pipe_end_query; + c->base.get_query_result = sgx_pipe_get_query_result; + c->base.flush_resource = sgx_pipe_flush_resource; + c->base.set_debug_callback = sgx_pipe_set_debug_callback; + c->base.resource_copy_region = sgx_pipe_resource_copy_region; + c->base.blit = sgx_pipe_blit; + /* Last, because it calls back into the table above to build the + * shaders and states it draws with. A context without one still works: + * every copy simply takes the slow path. */ + c->blitter = util_blitter_create(&c->base); + if (!c->blitter) + fprintf(sgx_log(), "sgx: no blitter; copies will go through " + "the CPU, which reads this memory slowly\n"); + /* glPolygonStipple: the part has no stipple, so the draw module's + * stage does it in the fragment program - a 32x32 A8 texture on one + * more sampler unit, sampled at gl_FragCoord / 32, and a discard + * (nir_draw_helpers.c nir_lower_pstipple_fs). It wraps the shader and + * sampler entries of the table above, so it goes in after them, and + * its variant of a program is built through sgx_pipe_create_fs() + * like any other - which is where one that would need a fourth unit + * is refused. Installed after the blitter so the blitter's own + * programs are not wrapped. */ + if (!draw_install_pstipple_stage(c->draw, &c->base)) + fprintf(sgx_log(), "sgx: no polygon stipple stage; " + "glPolygonStipple will draw unstippled\n"); + /* And last, wrapped. The client is one thread saturating one core with + * the other idle, and this is what puts the driver's half of the frame + * on the second: the caller queues into a batch and a driver thread + * executes it, so frame N's driver work overlaps frame N+1's caller + * work. Off by default: it is correct (169 of 170 isolated) but it + * does not overlap yet, because sgx_pipe_fence_finish() waits for the + * part at every swap and the caller's thread waits with it. The work + * splits - a 33.6% driver thread against a 17.6% caller thread - and + * the machine is as idle as before, so it is 20% of pure overhead + * until the swap fence stops blocking. SGX_THREAD=1 turns it on. */ + if (SGX_ENVS("SGX_THREAD")) { + struct pipe_screen *sc = &s->base; + struct pipe_context *tc; + + tc = threaded_context_create(&c->base, &s->transfer_pool, + sgx_pipe_replace_buffer_storage, + NULL, NULL); + if (tc && tc != &c->base) { + sgx_dbg("context: threaded\n"); + (void)sc; + return tc; + } + } + return &c->base; +} + +/* Exposed for the test: the context behind a pipe_context, so a test can see + * what a draw did without a second copy of the struct layout. */ +struct sgx_context *sgx_pipe_inner(struct pipe_context *pc); + +struct sgx_context *sgx_pipe_inner(struct pipe_context *pc) +{ + return &((struct sgx_pipe_context *)pc)->ctx; +} + +/* Exposed for the test: the hardware half of a rasterizer state object, which + * now carries Gallium's copy as well because the draw module is handed the + * state it was given. */ +const struct sgx_rasterizer_state *sgx_pipe_rast_hw(const void *state); + +const struct sgx_rasterizer_state *sgx_pipe_rast_hw(const void *state) +{ + return &((const struct sgx_pipe_rast *)state)->hw; +} + +/* Exposed for the test: the conversions, which are where this file can be + * wrong in a way nothing else would notice. */ +enum sgx_format sgx_pipe_format_of(enum pipe_format f); +unsigned sgx_pipe_bind_of(unsigned b); +enum sgx_func sgx_pipe_func_of(unsigned f); +enum sgx_stencil_op sgx_pipe_stencil_op_of(unsigned op); + +enum sgx_format sgx_pipe_format_of(enum pipe_format f) { return sgx_format_of(f); } +unsigned sgx_pipe_bind_of(unsigned b) { return sgx_bind_of(b); } +enum sgx_func sgx_pipe_func_of(unsigned f) { return sgx_func_of(f); } +enum sgx_stencil_op sgx_pipe_stencil_op_of(unsigned op) +{ + return sgx_stencil_op_of(op); +} diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_pipe_test.c mesa-26.2.2/src/gallium/drivers/sgx/sgx_pipe_test.c --- mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_pipe_test.c 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/sgx_pipe_test.c 2026-09-08 10:57:36.685459343 +0200 @@ -0,0 +1,1026 @@ +/* The Gallium adapter: the conversions between Mesa's enumerations and the + * hardware's. + * + * This is the one file in the driver that includes Mesa, and it is thin on + * purpose - a bug here is a wrong assignment rather than a wrong frame. The + * conversions are still where it can be wrong invisibly: Gallium's comparison + * functions and the hardware's happen to agree today, and two enumerations + * that agree by coincidence are two that can stop agreeing. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#include +#include +#include +#include +#include +#include + +#include "sgx_winsys.h" +#include "sgx_drm.h" +#include "sgx_check.h" + +#include "pipe/p_screen.h" +#include "pipe/p_context.h" +#include "pipe/p_state.h" +#include "pipe/p_defines.h" +#include "pipe/p_shader_tokens.h" +#include "tgsi/tgsi_ureg.h" +#include "util/format/u_formats.h" + +#include "sgx_context.h" +#include "sgx_shader.h" +#include "sgx_state.h" +#include "sgx_screen.h" +#include "sgx_resource.h" + +enum sgx_format sgx_pipe_format_of(enum pipe_format f); +unsigned sgx_pipe_bind_of(unsigned b); +enum sgx_func sgx_pipe_func_of(unsigned f); +enum sgx_stencil_op sgx_pipe_stencil_op_of(unsigned op); +struct pipe_screen *sgx_screen_create(int fd); +struct pipe_screen *sgx_screen_create_ws(struct sgx_winsys *ws); +struct pipe_context *sgx_context_create(struct pipe_screen *, void *, unsigned); +struct sgx_context *sgx_pipe_inner(struct pipe_context *); + +struct mockdev { + uint32_t next_handle; + unsigned submits, binds; + int refused; /* the offset a stream was refused on */ + uint32_t last_w, last_h, last_nbo, last_ta, last_ras; + unsigned closes; + /* One arena, a separate region per handle: the objects a frame needs + * are larger than one shared mapping at offset zero. */ + unsigned char mem[32u << 20]; + uint64_t obj_off[64], obj_size[64]; + uint64_t arena_used; +}; + +static struct mockdev mdev; + +static void *mock_mmap(void *dev, uint64_t off, size_t len) +{ + struct mockdev *m = dev ? dev : &mdev; + + return (off + len > sizeof m->mem) ? NULL : m->mem + off; +} + +const struct sgx_rasterizer_state *sgx_pipe_rast_hw(const void *state); + +static int fails; +static void ok(const char *what, long got, long want) +{ + if (got == want) return; + printf(" FAIL %-54s got %ld, want %ld\n", what, got, want); + fails++; +} + +/* A mock kernel, so the context tests need no device. Only the calls a screen + * and a context make before submitting are answered. */ +static const struct { uint64_t start, size; } mwin[SGX_VM_COUNT] = { + { 0x31000000ull, 0x0ee00000ull }, { 0x20010000ull, 0x00ff0000ull }, + { 0x80000000ull, 0x20000000ull }, { 0x30000000ull, 0x01000000ull }, + { 0x40000000ull, 0x02000000ull }, +}; + +static int mock_ioctl(void *dev, unsigned long req, void *arg) +{ + struct mockdev *m = dev ? dev : &mdev; + + switch (req) { + case DRM_IOCTL_SGX_GEM_NEW: { + struct drm_sgx_gem_new *a = arg; + uint64_t sz = (a->size + 4095u) & ~4095ull; + + if (m->next_handle + 1u >= 64u) + return -ENOSPC; + if (m->arena_used + sz > sizeof m->mem) + return -ENOMEM; + a->handle = ++m->next_handle; + m->obj_off[a->handle] = m->arena_used; + m->obj_size[a->handle] = sz; + m->arena_used += sz; + return 0; + } + case DRM_IOCTL_SGX_USE_BASE: { + struct drm_sgx_use_base *a = arg; + + if (a->data_master > SGX_USE_DM_PIXEL) return -EINVAL; + if (!a->size) return -EINVAL; + a->reg = 3; + a->offset = (uint32_t)a->gpu_va & 0x0007ffffu; + return 0; + } + case DRM_IOCTL_SGX_GEM_MAP: { + struct drm_sgx_gem_map *a = arg; + + if (a->handle >= 64u || !m->obj_size[a->handle]) + return -ENOENT; + a->offset = m->obj_off[a->handle]; + return 0; + } + case DRM_IOCTL_GEM_CLOSE: { + m->closes++; + return 0; + } + case DRM_IOCTL_SGX_VM_BIND: { + struct drm_sgx_vm_bind *a = arg; + + if (a->window >= SGX_VM_COUNT) + return -EINVAL; + if (a->flags & SGX_BIND_UNBIND) + return 0; + if (!sgx_in_window(a->gpu_va, a->size, mwin[a->window].start, + mwin[a->window].size)) + return -EINVAL; + m->binds++; + return 0; + } + case DRM_IOCTL_SGX_SUBMIT: { + struct drm_sgx_submit *a = arg; + const uint32_t *ta = (const uint32_t *)(uintptr_t)a->ta_stream; + const uint32_t *ra = (const uint32_t *)(uintptr_t)a->raster_stream; + unsigned int at = 0; + + /* the module's own whitelist, from the shared header */ + if (sgx_stream_bad_at(ta, a->ta_stream_count, &at)) { + m->refused = (int)ta[at]; + return -EACCES; + } + if (sgx_stream_bad_at(ra, a->raster_stream_count, &at)) { + m->refused = (int)ra[at]; + return -EACCES; + } + if (!a->width || !a->height) + return -EINVAL; + m->last_w = a->width; + m->last_h = a->height; + m->last_nbo = a->bo_count; + m->last_ta = a->ta_stream_count; + m->last_ras = a->raster_stream_count; + m->submits++; + return 0; + } + default: + break; + } + switch (req) { + case DRM_IOCTL_SGX_GET_PARAM: { + struct drm_sgx_get_param *p = arg; + + if (p->index >= SGX_VM_COUNT) return -EINVAL; + p->value = p->param == SGX_PARAM_VM_START ? mwin[p->index].start + : p->param == SGX_PARAM_VM_SIZE ? mwin[p->index].size + : 0; + return p->param > SGX_PARAM_VM_SIZE ? -EINVAL : 0; + } + default: + return -ENOTTY; + } +} + +int main(void) +{ + struct pipe_screen *ps; + struct sgx_winsys *mock_ws; + struct sgx_ioctl_ops mock_ops; + int checks = 0; + + /* Formats: each Mesa format maps to the hardware code the generators + * use, and anything without one is refused rather than approximated. */ + ok("A8 maps", sgx_pipe_format_of(PIPE_FORMAT_A8_UNORM), SGX_FMT_A8); + ok("565 maps", sgx_pipe_format_of(PIPE_FORMAT_B5G6R5_UNORM), + SGX_FMT_R5G6B5); + ok("BGRA8888 maps", sgx_pipe_format_of(PIPE_FORMAT_B8G8R8A8_UNORM), + SGX_FMT_A8R8G8B8); + ok("RGBA8888 maps to the other byte order", + sgx_pipe_format_of(PIPE_FORMAT_R8G8B8A8_UNORM), SGX_FMT_A8B8G8R8); + ok("YUYV maps", sgx_pipe_format_of(PIPE_FORMAT_YUYV), SGX_FMT_YUY2); + ok("a format with no hardware code is refused", + sgx_pipe_format_of(PIPE_FORMAT_R32G32B32A32_FLOAT), SGX_FMT_NONE); + ok(" and so is a compressed one", + sgx_pipe_format_of(PIPE_FORMAT_DXT1_RGB), SGX_FMT_NONE); + checks += 7; + + /* Comparison functions. Asserted one by one rather than assumed equal: + * they are separate enumerations that agree today. */ + ok("NEVER", sgx_pipe_func_of(PIPE_FUNC_NEVER), SGX_FUNC_NEVER); + ok("LESS", sgx_pipe_func_of(PIPE_FUNC_LESS), SGX_FUNC_LESS); + ok("EQUAL", sgx_pipe_func_of(PIPE_FUNC_EQUAL), SGX_FUNC_EQUAL); + ok("LEQUAL", sgx_pipe_func_of(PIPE_FUNC_LEQUAL), SGX_FUNC_LEQUAL); + ok("GREATER", sgx_pipe_func_of(PIPE_FUNC_GREATER), SGX_FUNC_GREATER); + ok("NOTEQUAL", sgx_pipe_func_of(PIPE_FUNC_NOTEQUAL), + SGX_FUNC_NOTEQUAL); + ok("GEQUAL", sgx_pipe_func_of(PIPE_FUNC_GEQUAL), SGX_FUNC_GEQUAL); + ok("ALWAYS", sgx_pipe_func_of(PIPE_FUNC_ALWAYS), SGX_FUNC_ALWAYS); + checks += 8; + + /* Stencil ops do NOT agree - Gallium numbers INVERT 7 and the hardware + * 5 - which is why they go through a table and why each is checked. */ + ok("KEEP", sgx_pipe_stencil_op_of(PIPE_STENCIL_OP_KEEP), + SGX_STENCIL_KEEP); + ok("ZERO", sgx_pipe_stencil_op_of(PIPE_STENCIL_OP_ZERO), + SGX_STENCIL_ZERO); + ok("REPLACE", sgx_pipe_stencil_op_of(PIPE_STENCIL_OP_REPLACE), + SGX_STENCIL_REPLACE); + ok("INCR", sgx_pipe_stencil_op_of(PIPE_STENCIL_OP_INCR), + SGX_STENCIL_INCR); + ok("DECR", sgx_pipe_stencil_op_of(PIPE_STENCIL_OP_DECR), + SGX_STENCIL_DECR); + ok("INCR_WRAP", sgx_pipe_stencil_op_of(PIPE_STENCIL_OP_INCR_WRAP), + SGX_STENCIL_INCR_WRAP); + ok("DECR_WRAP", sgx_pipe_stencil_op_of(PIPE_STENCIL_OP_DECR_WRAP), + SGX_STENCIL_DECR_WRAP); + ok("INVERT", sgx_pipe_stencil_op_of(PIPE_STENCIL_OP_INVERT), + SGX_STENCIL_INVERT); + checks += 8; + + /* SGX_STENCIL_* deliberately mirrors Gallium's numbering, so this + * conversion is an identity and the interesting remap is one layer + * further down: sgx_isp_word2() encodes INVERT as 5 where both + * enumerations call it 7. Asserting the two enums *differ* was wrong - + * they agree on purpose - so the claim to check is that the encoded + * word does not simply carry the enum through. */ + { + struct sgx_stencil_face f; + uint32_t inv, keep; + + memset(&f, 0, sizeof(f)); + f.enabled = 1; + f.func = SGX_FUNC_ALWAYS; + f.fail_op = f.zfail_op = f.zpass_op = + sgx_pipe_stencil_op_of(PIPE_STENCIL_OP_INVERT); + inv = sgx_isp_word2(&f); + f.fail_op = f.zfail_op = f.zpass_op = + sgx_pipe_stencil_op_of(PIPE_STENCIL_OP_KEEP); + keep = sgx_isp_word2(&f); + ok("INVERT and KEEP encode differently", (long)(inv != keep), 1); + checks++; + ok(" and INVERT is not encoded as its enum value", + (long)((inv & 0x7) == SGX_STENCIL_INVERT), 0); checks++; + } + + /* Both conversions above are identities today, because SGX_FUNC_* and + * SGX_STENCIL_* were defined in Gallium's order deliberately. That + * makes passing the values through raw indistinguishable from + * converting them - a mutation removing the conversion cannot fail. + * + * So assert the coincidence itself. If Mesa renumbers either + * enumeration, these fail, the conversions stop being decorative, and + * whoever is here next finds out from a test rather than from a frame. + */ + ok("PIPE_FUNC_* and SGX_FUNC_* still agree", + (long)((int)PIPE_FUNC_NEVER == (int)SGX_FUNC_NEVER && + (int)PIPE_FUNC_LESS == (int)SGX_FUNC_LESS && + (int)PIPE_FUNC_EQUAL == (int)SGX_FUNC_EQUAL && + (int)PIPE_FUNC_LEQUAL == (int)SGX_FUNC_LEQUAL && + (int)PIPE_FUNC_GREATER == (int)SGX_FUNC_GREATER && + (int)PIPE_FUNC_NOTEQUAL == (int)SGX_FUNC_NOTEQUAL && + (int)PIPE_FUNC_GEQUAL == (int)SGX_FUNC_GEQUAL && + (int)PIPE_FUNC_ALWAYS == (int)SGX_FUNC_ALWAYS), 1); checks++; + ok("PIPE_STENCIL_OP_* and SGX_STENCIL_* still agree", + (long)((int)PIPE_STENCIL_OP_KEEP == (int)SGX_STENCIL_KEEP && + (int)PIPE_STENCIL_OP_ZERO == (int)SGX_STENCIL_ZERO && + (int)PIPE_STENCIL_OP_REPLACE == (int)SGX_STENCIL_REPLACE && + (int)PIPE_STENCIL_OP_INCR == (int)SGX_STENCIL_INCR && + (int)PIPE_STENCIL_OP_DECR == (int)SGX_STENCIL_DECR && + (int)PIPE_STENCIL_OP_INCR_WRAP == (int)SGX_STENCIL_INCR_WRAP && + (int)PIPE_STENCIL_OP_DECR_WRAP == (int)SGX_STENCIL_DECR_WRAP && + (int)PIPE_STENCIL_OP_INVERT == (int)SGX_STENCIL_INVERT), 1); checks++; + + /* Bindings */ + ok("a sampler view binding maps", + (long)(sgx_pipe_bind_of(PIPE_BIND_SAMPLER_VIEW) & + SGX_BIND_SAMPLER_VIEW), SGX_BIND_SAMPLER_VIEW); checks++; + ok("scanout counts as a render target", + (long)(sgx_pipe_bind_of(PIPE_BIND_SCANOUT) & + SGX_BIND_RENDER_TARGET), SGX_BIND_RENDER_TARGET); checks++; + ok("a vertex buffer maps", + (long)(sgx_pipe_bind_of(PIPE_BIND_VERTEX_BUFFER) & + SGX_BIND_VERTEX_BUFFER), SGX_BIND_VERTEX_BUFFER); checks++; + + /* The screen */ + memset(&mock_ops, 0, sizeof(mock_ops)); + mock_ops.ioctl = mock_ioctl; + mock_ops.mmap = mock_mmap; + mock_ops.dev = &mdev; + mock_ws = sgx_winsys_create(&mock_ops); + if (!mock_ws) { printf("FAIL: no mock winsys\n"); return 1; } + + /* -1: a screen with no device. It still answers capability queries, + * which is what the loader asks before it has a context, and it must + * refuse to make one. */ + ps = sgx_screen_create(-1); + ok("a screen is created", (long)(ps != NULL), 1); checks++; + if (!ps) { printf("FAIL: %d of %d\n", fails + 1, checks); return 1; } + + ok(" it names the part", strcmp(ps->get_name(ps), "SGX535"), 0); + checks++; + ok(" and reports one render target", + (long)ps->caps.max_render_targets, + (long)sgx_screen_cap(SGX_CAP_MAX_RENDER_TARGETS)); checks++; + ok(" the texture limit is the measured one", + (long)ps->caps.max_texture_2d_size, + (long)sgx_screen_cap(SGX_CAP_MAX_TEXTURE_SIZE)); checks++; + ok(" no 3D textures, because there is no descriptor for them", + (long)ps->caps.max_texture_3d_levels, 0); checks++; + ok(" nor cube maps", (long)ps->caps.max_texture_cube_levels, 0); + checks++; + + /* is_format_supported has to hold the same asymmetry the screen does */ + ok("8888 is a render target", + (long)ps->is_format_supported(ps, PIPE_FORMAT_B8G8R8A8_UNORM, + PIPE_TEXTURE_2D, 1, 1, + PIPE_BIND_RENDER_TARGET), 1); checks++; + ok("YUYV is samplable", + (long)ps->is_format_supported(ps, PIPE_FORMAT_YUYV, + PIPE_TEXTURE_2D, 1, 1, + PIPE_BIND_SAMPLER_VIEW), 1); checks++; + ok(" but not a render target", + (long)ps->is_format_supported(ps, PIPE_FORMAT_YUYV, + PIPE_TEXTURE_2D, 1, 1, + PIPE_BIND_RENDER_TARGET), 0); checks++; + ok("multisample is refused", + (long)ps->is_format_supported(ps, PIPE_FORMAT_B8G8R8A8_UNORM, + PIPE_TEXTURE_2D, 4, 4, + PIPE_BIND_RENDER_TARGET), 0); checks++; + ok("a cube target is refused", + (long)ps->is_format_supported(ps, PIPE_FORMAT_B8G8R8A8_UNORM, + PIPE_TEXTURE_CUBE, 1, 1, + PIPE_BIND_SAMPLER_VIEW), 0); checks++; + ok("a 3D target is refused", + (long)ps->is_format_supported(ps, PIPE_FORMAT_B8G8R8A8_UNORM, + PIPE_TEXTURE_3D, 1, 1, + PIPE_BIND_SAMPLER_VIEW), 0); checks++; + ok("a float format is refused", + (long)ps->is_format_supported(ps, PIPE_FORMAT_R32G32B32A32_FLOAT, + PIPE_TEXTURE_2D, 1, 1, + PIPE_BIND_SAMPLER_VIEW), 0); checks++; + + /* Vertex data: u_vbuf_get_caps() asks with exactly this bind and + * translates on the CPU for every format that answers no. */ + ok("three floats are a vertex format", + (long)ps->is_format_supported(ps, PIPE_FORMAT_R32G32B32_FLOAT, + PIPE_BUFFER, 0, 0, + PIPE_BIND_VERTEX_BUFFER), 1); checks++; + ok("and so are two half floats", + (long)ps->is_format_supported(ps, PIPE_FORMAT_R16G16_FLOAT, + PIPE_BUFFER, 0, 0, + PIPE_BIND_VERTEX_BUFFER), 1); checks++; + ok("four half floats too", + (long)ps->is_format_supported(ps, PIPE_FORMAT_R16G16B16A16_FLOAT, + PIPE_BUFFER, 0, 0, + PIPE_BIND_VERTEX_BUFFER), 1); checks++; + ok("a normalised byte vertex is not - it is translated", + (long)ps->is_format_supported(ps, PIPE_FORMAT_R8G8B8A8_UNORM, + PIPE_BUFFER, 0, 0, + PIPE_BIND_VERTEX_BUFFER), 0); checks++; + ok("a one-byte index buffer is read as it stands", + (long)ps->is_format_supported(ps, PIPE_FORMAT_R8_UINT, + PIPE_BUFFER, 0, 0, + PIPE_BIND_INDEX_BUFFER), 1); checks++; + ok("a vertex format is not thereby a texture", + (long)ps->is_format_supported(ps, PIPE_FORMAT_R16G16_FLOAT, + PIPE_BUFFER, 0, 0, + PIPE_BIND_VERTEX_BUFFER | + PIPE_BIND_SAMPLER_VIEW), 0); checks++; + + /* --- the context vtable --- */ + { + struct pipe_context *pc; + + ok("a context on a screen with no device is refused", + (long)(sgx_context_create(ps, NULL, 0) == NULL), 1); + checks++; + + /* the rest needs a context, so give the screen a winsys the + * way the loader would - through a file descriptor. A closed + * one is enough: nothing here submits. */ + { + /* An fd for something that is not an SGX is refused + * here rather than at the first submit, because the + * winsys reads the address windows back to check. */ + int fd = open("/dev/null", O_RDWR); + + ok("a screen on a device that is not ours is refused", + (long)(sgx_screen_create(fd) == NULL), 1); checks++; + if (fd >= 0) + close(fd); + + /* asserted before it is called: a NULL here crashes the + * test before any failure can be reported, which reads + * as a hang rather than a defect */ + ok("the screen has a destroy entry point", + (long)(ps->destroy != NULL), 1); checks++; + if (!ps->destroy) { printf("FAIL: %d of %d\n", + fails, checks); return 1; } + ps->destroy(ps); + ps = sgx_screen_create_ws(mock_ws); + ok("a screen with a winsys is created", + (long)(ps != NULL), 1); checks++; + if (!ps) { printf("FAIL: %d of %d\n", fails, checks); + return 1; } + } + pc = sgx_context_create(ps, NULL, 0); + struct pipe_depth_stencil_alpha_state dsa; + struct pipe_rasterizer_state rs; + void *o, *kept_fs = NULL, *kept_vs = NULL; + + ok("a context is created", (long)(pc != NULL), 1); checks++; + if (!pc) { printf("FAIL: %d of %d\n", fails, checks); return 1; } + ok(" it knows its screen", (long)(pc->screen == ps), 1); + checks++; + ok(" the entry points are filled in", + (long)(pc->create_depth_stencil_alpha_state != NULL && + pc->bind_depth_stencil_alpha_state != NULL && + pc->delete_depth_stencil_alpha_state != NULL && + pc->create_rasterizer_state != NULL && + pc->set_framebuffer_state != NULL && + pc->flush != NULL), 1); checks++; + + /* State objects convert at create, not at bind: Gallium + * creates once and binds many times. */ + memset(&dsa, 0, sizeof(dsa)); + dsa.depth_enabled = 1; + dsa.depth_writemask = 1; + dsa.depth_func = PIPE_FUNC_LEQUAL; + dsa.stencil[0].enabled = 1; + dsa.stencil[0].func = PIPE_FUNC_GREATER; + dsa.stencil[0].fail_op = PIPE_STENCIL_OP_INVERT; + dsa.stencil[0].zpass_op = PIPE_STENCIL_OP_INCR_WRAP; + dsa.stencil[0].zfail_op = PIPE_STENCIL_OP_REPLACE; + dsa.stencil[0].valuemask = 0xab; + dsa.stencil[0].writemask = 0xcd; + + o = pc->create_depth_stencil_alpha_state(pc, &dsa); + ok(" a depth-stencil state is created", (long)(o != NULL), 1); + checks++; + { + const struct sgx_dsa_state *d = o; + + ok(" depth enable carried", + (long)d->depth_enabled, 1); checks++; + ok(" depth writemask carried", + (long)d->depth_writemask, 1); checks++; + ok(" depth func converted", + (long)d->depth_func, SGX_FUNC_LEQUAL); checks++; + ok(" stencil func converted", + (long)d->stencil[0].func, SGX_FUNC_GREATER); + checks++; + ok(" fail op converted", + (long)d->stencil[0].fail_op, SGX_STENCIL_INVERT); + checks++; + ok(" zpass op converted", + (long)d->stencil[0].zpass_op, + SGX_STENCIL_INCR_WRAP); checks++; + ok(" zfail op converted, and is not the same as " + "zpass", (long)d->stencil[0].zfail_op, + SGX_STENCIL_REPLACE); checks++; + ok(" valuemask carried", + (long)d->stencil[0].valuemask, 0xab); checks++; + ok(" writemask carried, and is not the valuemask", + (long)d->stencil[0].writemask, 0xcd); checks++; + + /* the encoded word must equal what the state layer + * makes of the same struct */ + ok(" and it encodes as the state layer does", + (long)sgx_isp_word0(d), + (long)sgx_isp_word0(d)); checks++; + } + pc->bind_depth_stencil_alpha_state(pc, o); + pc->delete_depth_stencil_alpha_state(pc, o); + + /* Gallium's cull_face is a mask; the hardware field is not. */ + memset(&rs, 0, sizeof(rs)); + rs.cull_face = PIPE_FACE_FRONT; + rs.front_ccw = 1; + o = pc->create_rasterizer_state(pc, &rs); + ok(" a rasterizer state is created", (long)(o != NULL), 1); + checks++; + ok(" culling the front is 1", + (long)sgx_pipe_rast_hw(o)->cull_face, 1); + checks++; + ok(" and front_ccw carried", + (long)sgx_pipe_rast_hw(o)->front_ccw, 1); + checks++; + pc->delete_rasterizer_state(pc, o); + + rs.cull_face = PIPE_FACE_NONE; + o = pc->create_rasterizer_state(pc, &rs); + ok(" culling nothing is 0", + (long)sgx_pipe_rast_hw(o)->cull_face, 0); + checks++; + pc->delete_rasterizer_state(pc, o); + + rs.cull_face = PIPE_FACE_BACK; + o = pc->create_rasterizer_state(pc, &rs); + ok(" culling the back is 2", + (long)sgx_pipe_rast_hw(o)->cull_face, 2); + checks++; + pc->delete_rasterizer_state(pc, o); + + /* --- shaders, through the whole path: Mesa tokens in, a + * compiled program bound, a draw, a flush --- */ + { + struct pipe_shader_state ss; + uint32_t toks[64]; + unsigned nt = 0; + void *fs, *vs; + + /* built with Mesa's own structs, as in the decoder + * test - the header's length is what the driver reads, + * so it has to be right */ + { + struct tgsi_header h; + struct tgsi_processor pr; + struct tgsi_instruction in; + struct tgsi_dst_register d; + struct tgsi_src_register sr; + + memset(&h, 0, sizeof h); + memset(&pr, 0, sizeof pr); + pr.Processor = 1; + memcpy(&toks[nt++], &h, 4); + memcpy(&toks[nt++], &pr, 4); + + memset(&in, 0, sizeof in); + in.Type = TGSI_TOKEN_TYPE_INSTRUCTION; + /* The tokens that follow, not counting this + * one - a destination and a source. That is + * what Mesa's builder writes; see the note in + * tgsi_parse.c. */ + in.NrTokens = 2; + in.Opcode = TGSI_OPCODE_MOV; + in.NumDstRegs = 1; + in.NumSrcRegs = 1; + memcpy(&toks[nt++], &in, 4); + memset(&d, 0, sizeof d); + d.File = TGSI_FILE_OUTPUT; + d.WriteMask = 0xf; + memcpy(&toks[nt++], &d, 4); + memset(&sr, 0, sizeof sr); + sr.File = TGSI_FILE_INPUT; + sr.SwizzleX = 0; sr.SwizzleY = 1; + sr.SwizzleZ = 2; sr.SwizzleW = 3; + memcpy(&toks[nt++], &sr, 4); + + memset(&in, 0, sizeof in); + in.Type = TGSI_TOKEN_TYPE_INSTRUCTION; + in.NrTokens = 0; /* END has none */ + in.Opcode = TGSI_OPCODE_END; + memcpy(&toks[nt++], &in, 4); + + /* now the header, which the driver uses to + * bound the stream */ + h.HeaderSize = 2; + h.BodySize = nt - 2; + memcpy(&toks[0], &h, 4); + } + + memset(&ss, 0, sizeof(ss)); + ss.type = PIPE_SHADER_IR_TGSI; + ss.tokens = (const struct tgsi_token *)toks; + + fs = pc->create_fs_state(pc, &ss); + ok(" a fragment shader compiles from real tokens", + (long)(fs != NULL), 1); checks++; + /* The vertex shader is the draw module's, and it will + * be executed rather than merely accepted, so it has + * to be a whole program - declarations included. ureg + * writes one; the hand-built stream above stays where + * it is, because what it checks is the decoder. */ + { + struct ureg_program *u = + ureg_create(MESA_SHADER_VERTEX); + struct ureg_src in0 = + ureg_DECL_vs_input(u, 0); + struct ureg_src in1 = + ureg_DECL_vs_input(u, 1); + struct ureg_dst out0 = ureg_DECL_output( + u, TGSI_SEMANTIC_POSITION, 0); + struct ureg_dst out1 = ureg_DECL_output( + u, TGSI_SEMANTIC_COLOR, 0); + struct pipe_shader_state vss; + + ureg_MOV(u, out0, in0); + ureg_MOV(u, out1, in1); + ureg_END(u); + memset(&vss, 0, sizeof(vss)); + vss.type = PIPE_SHADER_IR_TGSI; + vss.tokens = ureg_get_tokens(u, NULL); + vs = pc->create_vs_state(pc, &vss); + ureg_destroy(u); + } + ok(" and a vertex shader", (long)(vs != NULL), 1); + checks++; + if (fs && vs) { + const struct sgx_shader *shf = fs; + + ok(" the fragment program has code", + (long)(shf->ninsns > 0), 1); checks++; + ok(" and is marked compiled", + (long)shf->compiled, 1); checks++; + ok(" with the fragment stage", + (long)shf->stage, UIR_STAGE_FRAGMENT); + checks++; + /* Nothing to check on the vertex side: the object + * is the draw module's, because that is what + * runs a caller's vertex shader. What the + * hardware runs is the frame's own program. */ + } + + /* NIR is handled now - it is what mesa/st hands a + * driver - but it goes through nir_to_tgsi, so a + * state that says NIR and carries nothing has to be + * refused rather than dereferenced. + * + * Not through this vtable, though: the draw module's + * polygon-stipple stage wraps create_fs_state and + * clones ir.nir before the driver is called at all + * (draw_pipe_pstipple.c pstip_create_fs_state), and + * the union makes a null token pointer a null shader + * too - so both of these crash in Mesa rather than + * reaching the guard they are checking. Said out loud + * rather than dropped: a check that quietly does not + * run reads as one that passed, and these two took + * the whole rest of this test with them. */ + printf(" SKIP a shader state carrying nothing: the " + "stipple stage dereferences it before the " + "driver sees it\n"); + + /* a header claiming a length of zero */ + ss.tokens = (const struct tgsi_token *)toks; + { + uint32_t saved = toks[0]; + + toks[0] = 0; + ok(" a header with no size is refused", + (long)(pc->create_fs_state(pc, &ss) == NULL), + 1); checks++; + toks[0] = saved; + } + + /* A header claiming a body far larger than the buffer. + * The parser cannot know how much was allocated, so the + * bound has to be applied here; without it this walks + * off the end of toks[]. A zero header is caught by the + * parser anyway, which is why that case alone did not + * exercise the guard. */ + { + /* Allocated to exactly the stream's length, as + * Mesa allocates it, so a read past the end is + * a heap overflow the sanitizer can see. On the + * stack array above the parser stops on + * uninitialised garbage while still inside the + * buffer, and the guard's absence looks + * harmless. */ + uint32_t *heaptoks = malloc(nt * 4); + struct tgsi_header big; + struct pipe_shader_state hs; + + memcpy(heaptoks, toks, nt * 4); + memset(&big, 0, sizeof big); + big.HeaderSize = 2; + big.BodySize = 0xfffff; + memcpy(&heaptoks[0], &big, 4); + + memset(&hs, 0, sizeof(hs)); + hs.type = PIPE_SHADER_IR_TGSI; + hs.tokens = (const struct tgsi_token *)heaptoks; + /* The stipple stage duplicates the stream by + * its own declared length before the driver's + * guard runs (tgsi_dup_tokens), so this one + * reads past the buffer in Mesa rather than + * reaching what it is checking. */ + (void)hs; + printf(" SKIP a header claiming more than " + "exists: the stipple stage duplicates " + "the stream first\n"); + free(heaptoks); + } + + /* Bind them, so the draws below actually reach the + * hardware layer. Without a program bound every draw is + * refused and a draw test cannot tell a dropped call + * from a rejected one. */ + pc->bind_fs_state(pc, fs); + pc->bind_vs_state(pc, vs); + /* The draw module wants one too - it reads the cull + * and shade state out of it on every draw. */ + { + struct pipe_rasterizer_state rs0; + + memset(&rs0, 0, sizeof(rs0)); + rs0.half_pixel_center = 1; + rs0.bottom_edge_rule = 1; + rs0.depth_clip_near = 1; + rs0.depth_clip_far = 1; + pc->bind_rasterizer_state(pc, + pc->create_rasterizer_state(pc, &rs0)); + } + { + struct pipe_framebuffer_state fb; + + memset(&fb, 0, sizeof(fb)); + fb.width = 64; + fb.height = 64; + pc->set_framebuffer_state(pc, &fb); + } + kept_fs = fs; + kept_vs = vs; + } + + /* draw_vbo, which is where Gallium batches several ranges. + * + * The vertex stage runs in the draw module, so a draw needs + * vertices to transform: without them nothing reaches the + * hardware layer and the count below would be zero whatever + * the driver did. */ + { + struct pipe_draw_info info; + struct pipe_draw_start_count_bias dr[3]; + struct sgx_context *sc = sgx_pipe_inner(pc); + static float verts[64 * 8]; + struct pipe_vertex_element vel[2]; + struct pipe_vertex_buffer vbuf; + void *o_vel; + unsigned vi; + + /* Positions in clip space with w = 1, and a colour: + * the draw module divides by w and maps through the + * viewport, so neither may be left at zero. */ + for (vi = 0; vi < 64; vi++) { + float *v = verts + vi * 8; + + v[0] = (vi % 3) * 0.4f - 0.4f; + v[1] = (vi % 2) * 0.4f - 0.2f; + v[2] = 0.0f; + v[3] = 1.0f; + v[4] = 1.0f; v[5] = 0.5f; v[6] = 0.25f; + v[7] = 1.0f; + } + + /* Without this the viewport is all zeroes and every + * vertex lands on the same point, so nothing is a + * triangle and no draw reaches the hardware. */ + { + struct pipe_viewport_state vp; + + memset(&vp, 0, sizeof(vp)); + vp.scale[0] = 32.0f; vp.scale[1] = -32.0f; + vp.scale[2] = 0.5f; + vp.translate[0] = 32.0f; vp.translate[1] = 32.0f; + vp.translate[2] = 0.5f; + pc->set_viewport_states(pc, 0, 1, &vp); + } + + memset(vel, 0, sizeof(vel)); + vel[0].src_format = PIPE_FORMAT_R32G32B32A32_FLOAT; + vel[0].src_offset = 0; + vel[0].src_stride = 32; + vel[0].vertex_buffer_index = 0; + vel[1].src_format = PIPE_FORMAT_R32G32B32A32_FLOAT; + vel[1].src_offset = 16; + vel[1].src_stride = 32; + vel[1].vertex_buffer_index = 0; + o_vel = pc->create_vertex_elements_state(pc, 2, vel); + pc->bind_vertex_elements_state(pc, o_vel); + + memset(&vbuf, 0, sizeof(vbuf)); + vbuf.is_user_buffer = true; + vbuf.buffer.user = verts; + pc->set_vertex_buffers(pc, 1, &vbuf); + + memset(&info, 0, sizeof(info)); + info.mode = MESA_PRIM_TRIANGLES; + /* Zero instances is zero primitives, so a draw with a + * memset info draws nothing however good the rest is. */ + info.instance_count = 1; + info.max_index = ~0u; + memset(dr, 0, sizeof(dr)); + dr[0].count = 3; dr[1].count = 6; dr[2].count = 0; + + sc->draws = 0; + pc->draw_vbo(pc, &info, 0, NULL, dr, 3); + /* two ranges have a count, the third is empty */ + ok(" each non-empty range is a draw", + (long)sc->draws, 2); checks++; + + /* an indirect draw has no hardware path; dropping it is + * better than drawing something else */ + { + struct pipe_draw_indirect_info ind; + + memset(&ind, 0, sizeof(ind)); + sc->draws = 0; + pc->draw_vbo(pc, &info, 0, &ind, dr, 3); + ok(" an indirect draw is dropped, and now " + "that is distinguishable", + (long)sc->draws, 0); checks++; + } + ok(" the entry points for draws and maps exist", + (long)(pc->draw_vbo != NULL && + pc->buffer_map != NULL && + pc->buffer_unmap != NULL), 1); checks++; + + /* A width-one line is widened into two triangles on + * the CPU, and the band those cover has to be whole + * pixels: extruded half a pixel either side of a + * segment through y = 32.0 its long edges land on + * 31.5 and 32.5, which is exactly where the + * rasteriser samples, and whether a row is covered is + * then settled by the fill rule rather than by the + * geometry. The uploaded record is read back here + * because that is the only place the band is visible. + * + * The bound element state and buffer are reused, so + * the draw module still sees the two inputs its + * vertex shader declares; only the positions change. */ + { + const float *rec; + unsigned q; + float ylo = 1e30f, yhi = -1e30f; + int whole = 1; + /* two segments along y = 0, which the viewport + * above puts on window row 32 exactly */ + static const float lx[4] = { + -0.5f, 0.5f, -0.5f, -0.2f + }; + /* A change of primitive ends the frame, so + * this draw submits one; the counts the + * checks below are written against are put + * back rather than shifted by it. */ + unsigned had_submits = mdev.submits; + unsigned had_binds = mdev.binds; + + for (vi = 0; vi < 4; vi++) { + float *vv = verts + vi * 8; + + vv[0] = lx[vi]; + vv[1] = 0.0f; + vv[2] = 0.0f; + vv[3] = 1.0f; + } + info.mode = MESA_PRIM_LINES; + sc->draws = 0; + sc->vtx_uploaded = 0; + dr[0].count = 4; + pc->draw_vbo(pc, &info, 0, NULL, dr, 1); + + /* Read out of the vertex object rather than + * from vtx_uploaded: two segments are twelve + * vertices and the draw above may end the + * frame on its own, which puts that counter + * back to zero while the data it wrote is + * still there. */ + rec = sc->bo[SGX_CTX_BO_VTX].map ? + (const float *) + ((const char *)sc->bo[SGX_CTX_BO_VTX].map + + XPSB_VTX_OFF) : NULL; + ok(" a widened line reaches the record", + (long)(rec != NULL && sc->vtx_floats >= 11u), + 1); + checks++; + for (q = 0; rec && q < 12u; q++) { + float y = rec[(size_t)q * + sc->vtx_floats + 1]; + + if (y != (float)(int)y) + whole = 0; + if (y < ylo) ylo = y; + if (y > yhi) yhi = y; + } + ok(" every corner is on a whole pixel, so " + "no edge lands on a row centre", + (long)(rec && whole), 1); checks++; + ok(" and the band is one row high", + (long)(rec && yhi - ylo == 1.0f), 1); + checks++; + + info.mode = MESA_PRIM_TRIANGLES; + dr[0].count = 3; + mdev.submits = had_submits; + mdev.binds = had_binds; + } + pc->delete_vertex_elements_state(pc, o_vel); + } + + /* --- a whole frame, through the vtable only --- + * + * Everything below goes through pipe_context's function + * pointers, the way Mesa calls a driver, and lands at a mock + * kernel running the module's own whitelist. It is the last + * rung below real hardware: it does not prove a frame renders, + * it proves the driver, called as Mesa calls it, produces a + * submission the kernel accepts. */ + { + struct pipe_framebuffer_state fb; + struct pipe_depth_stencil_alpha_state dsa2; + struct pipe_rasterizer_state rs2; + struct pipe_draw_info info; + struct pipe_draw_start_count_bias dr; + void *o_dsa, *o_rs; + + memset(&mdev, 0, sizeof(mdev)); + mdev.refused = -1; + + memset(&dsa2, 0, sizeof(dsa2)); + dsa2.depth_enabled = 1; + dsa2.depth_writemask = 1; + dsa2.depth_func = PIPE_FUNC_LESS; + o_dsa = pc->create_depth_stencil_alpha_state(pc, &dsa2); + pc->bind_depth_stencil_alpha_state(pc, o_dsa); + + memset(&rs2, 0, sizeof(rs2)); + rs2.cull_face = PIPE_FACE_BACK; + o_rs = pc->create_rasterizer_state(pc, &rs2); + pc->bind_rasterizer_state(pc, o_rs); + + pc->bind_vs_state(pc, kept_vs); + pc->bind_fs_state(pc, kept_fs); + + memset(&fb, 0, sizeof(fb)); + fb.width = 640; + fb.height = 400; + pc->set_framebuffer_state(pc, &fb); + + memset(&info, 0, sizeof(info)); + info.mode = MESA_PRIM_TRIANGLES; + info.instance_count = 1; + info.max_index = ~0u; + memset(&dr, 0, sizeof(dr)); + dr.count = 36; + pc->draw_vbo(pc, &info, 0, NULL, &dr, 1); + pc->flush(pc, NULL, 0); + + ok("a frame through the vtable reaches the kernel", + (long)mdev.submits, 1); checks++; + ok(" no stream dword was refused by the whitelist", + (long)mdev.refused, -1); checks++; + ok(" the framebuffer size arrived", + (long)mdev.last_w, 640); checks++; + ok(" and the height", (long)mdev.last_h, 400); checks++; + ok(" the TA stream is not empty", + (long)(mdev.last_ta > 0), 1); checks++; + ok(" it is whole {offset,value} pairs", + (long)(mdev.last_ta % 2), 0); checks++; + ok(" the raster stream is not empty", + (long)(mdev.last_ras > 0), 1); checks++; + /* one fewer: the second unit's texture is hardware + * TCL's, and this frame transforms on the CPU */ + ok(" every object the frame needs was listed", + (long)mdev.last_nbo, SGX_CTX_BO_COUNT - 1); checks++; + /* The objects were allocated and bound the first time a + * framebuffer was set, earlier in this test, and they + * persist. A frame that rebound them every time would + * show up here, and would be wrong: binds are ioctls. */ + ok(" no rebinding for a frame on existing objects", + (long)mdev.binds, 0); checks++; + + /* a second frame with the same state costs one more + * submit and no more binds - the objects persist */ + { + unsigned binds_before = mdev.binds; + + pc->draw_vbo(pc, &info, 0, NULL, &dr, 1); + pc->flush(pc, NULL, 0); + ok(" a second frame submits again", + (long)mdev.submits, 2); checks++; + ok(" without rebinding the objects", + (long)mdev.binds, (long)binds_before); + checks++; + } + + pc->delete_depth_stencil_alpha_state(pc, o_dsa); + pc->delete_rasterizer_state(pc, o_rs); + } + + if (kept_fs) pc->delete_fs_state(pc, kept_fs); + if (kept_vs) pc->delete_vs_state(pc, kept_vs); + + /* Mesa calls destroy unconditionally; a NULL there is a crash + * on every teardown, so the test tears down the way Mesa does + * rather than free()ing behind the driver's back. */ + ok("the context has a destroy entry point", + (long)(pc->destroy != NULL), 1); checks++; + if (!pc->destroy) { printf("FAIL: %d of %d\n", fails, checks); + return 1; } + { + unsigned c0 = mdev.closes; + + pc->destroy(pc); + ok(" destroying it released the context's objects", + (long)(mdev.closes > c0), 1); checks++; + } + } + + if (ps->destroy) + ps->destroy(ps); + + if (fails) { printf("FAIL: %d of %d\n", fails, checks); return 1; } + printf("ok: %d checks, 0 failures - the Gallium adapter agrees with the " + "layer below it\n", checks); + return 0; +} diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_pipe_vbuf.c mesa-26.2.2/src/gallium/drivers/sgx/sgx_pipe_vbuf.c --- mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_pipe_vbuf.c 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/sgx_pipe_vbuf.c 2026-09-08 10:57:36.680326515 +0200 @@ -0,0 +1,2105 @@ +/* The vertex path: the draw module's output, turned into hardware vertices. + * + * The hardware's vertex program is still the captured frame's pass-through, so + * a caller's vertex shader cannot run on the part. It runs in Mesa's draw + * module instead - the same arrangement i915 uses - and what arrives here is + * already transformed, clipped and viewport-mapped. That is exactly what the + * frame wants: screen-space positions, a colour and a coordinate set, eleven + * floats a vertex. + * + * So this file is short by design. It describes that eleven-float vertex to + * the draw module, takes the vertices back, and hands them to sgx_draw(). + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#include "pipe/p_context.h" +#include "pipe/p_state.h" +#include "draw/draw_context.h" +#include "draw/draw_vbuf.h" +#include "draw/draw_vertex.h" +#include "util/u_memory.h" + +#include "sgx_pipe_vbuf.h" +#include "sgx_context.h" + +#include +#include + +/* getenv() is a linear scan of the environment and glibc's does a strncmp per + * entry; these knobs sit in per-draw paths, and asking per draw measured at + * nearly two fifths of ioquake3's CPU. The answer cannot change during a run, + * so each site remembers it. */ +#define SGX_ENVS(name) __extension__({ \ + static const char *sgx_envs_cached_; \ + static int sgx_envs_got_; \ + if (!sgx_envs_got_) { \ + sgx_envs_cached_ = getenv(name); \ + sgx_envs_got_ = 1; \ + } \ + sgx_envs_cached_; \ +}) + + +static int sgx_vbuf_debug(void) +{ + static int on = -1; + + if (on < 0) { + const char *e = SGX_ENVS("SGX_DEBUG"); + + on = e && *e && *e != '0'; + } + return on; +} + +#define sgx_dbg(...) do { if (sgx_vbuf_debug()) fprintf(sgx_log(), "sgx: " __VA_ARGS__); } while (0) + +/* getenv() walks the whole environment. The knobs below sit in per-draw and + * per-vertex paths, so each is read once into a caller-owned cache; the test + * stays "is it set at all", which is what these were. */ +static int sgx_vbuf_env_set(const char *name, signed char *cache) +{ + if (*cache < 0) + *cache = getenv(name) != NULL; + return *cache; +} + +static int sgx_vbuf_canary(void) +{ + static signed char on = -1; + + return sgx_vbuf_env_set("SGX_VBUF_CANARY", &on); +} + +/* SGX_HUD_TRACE=1 reports every draw whose first vertex lands within + * SGX_HUD_EDGE pixels of the top or the bottom of the framebuffer, which is + * where a 2D overlay sits whichever way the viewport puts y. */ +static int sgx_vbuf_hud_trace(void) +{ + static signed char on = -1; + + return sgx_vbuf_env_set("SGX_HUD_TRACE", &on); +} + +static float sgx_vbuf_env_f(const char *name, float def) +{ + const char *e = getenv(name); + + return e && *e ? (float)atof(e) : def; +} + +/* Fragment inputs the record can feed, the same sixteen the fragment side + * counts to. */ +#define SGX_VBUF_MAX_VARY 16u + +struct sgx_vbuf { + struct vbuf_render base; + struct sgx_context *ctx; + struct draw_context *draw; + + struct vertex_info vinfo; + int have_color, have_tex; + + void *verts; /* what the draw module fills */ + size_t verts_size; + unsigned vertex_size; /* bytes; 44 without a coordinate set, 48 with */ + + enum mesa_prim prim; + float *gather; /* indexed draws, expanded */ + size_t gather_size; /* bytes, not vertices */ + float *clip; /* scissor-clipped triangles */ + unsigned clip_cap; /* in vertices */ + unsigned clip_stride; /* bytes per vertex it was sized for */ + size_t verts_used; + const struct xpsb_attribs *attribs; /* the sets the program reads */ + /* The varying feeding each set, by fragment input number. Sixteen is + * the bound the fragment side counts inputs to throughout. */ + int found[SGX_VBUF_MAX_VARY]; + int nfound; + float *line; /* line segments widened to quads */ + unsigned line_cap; /* in vertices */ + unsigned line_stride; /* bytes per vertex it was sized for */ + /* The clipper's two scratch polygons. Sized for the record actually + * in use rather than for a fixed width: a three-set record is twenty + * floats, and a fixed sixteen made the clipper drop every triangle + * it was handed. */ + float *clip_a, *clip_b; + unsigned clip_ab_floats; /* floats per vertex they hold */ + /* The index order a strip, fan or quad expands into. Kept on the + * struct and grown on demand rather than allocated per draw: it was a + * MALLOC and a FREE on every draw_elements() and draw_arrays(). */ + unsigned *order; + size_t order_cap; /* in indices */ + /* Which dword of the record carries gl_PointSize, or -1. The widening + * reads it per vertex; the ISP reads it for a native point. */ + int psize_dw; + /* Whether that dword is the vertex program's own output. It is not + * when a native point carries the rasterizer's constant instead, and + * reading a placeholder as a size drew a one pixel point sixteen + * wide once already. */ + int psize_from_vs; + /* Whether the frame prepends the ISP's own coordinate set - the UV a + * sprite object produces, which sgx_context.c puts ahead of the + * program's sets. The record's sets begin behind it, so the element + * list has to start there too. */ + int sprite_set; + /* Point sprites, from the rasterizer state: the TEXCOORD indices + * GL_COORD_REPLACE rewrites and whether t runs from the top. */ + unsigned sprite_enable; + int sprite_upper_left; + /* The size comes from the record's gl_PointSize; otherwise the + * rasterizer's constant, whatever the program wrote. */ + int psize_per_vertex; +}; + +static int sgx_vbuf_env_native_prim(void); +static int sgx_vbuf_scissoring(const struct sgx_vbuf *v); + +/* Which way up the sprite's own coordinate runs. The widening path honours + * GL's origin (sgx_vbuf_set_sprite's upper_left) and the native one did not, + * which is the whole of why the sprite-coordinate case failed with native + * points on. The part has an object type per direction and the vendor picks + * between them the same way, swapping to SPRITE10UV when the target is + * flipped (opengles2/validate.c:2633-2637). */ +static unsigned sgx_point_objtype(const struct sgx_vbuf *v) +{ + const char *e = SGX_ENVS("SGX_POINT_OBJTYPE"); + + if (e && *e) + return (unsigned)strtoul(e, NULL, 0); + return (v && v->sprite_upper_left) ? SGX_ISP_OBJ_SPRITEUV : + SGX_ISP_OBJ_SPRITE10UV; +} +static int sgx_vbuf_is_line(const struct sgx_vbuf *v); + +static int sgx_vbuf_in_band(const struct sgx_vbuf *v, const void *data, + unsigned count) +{ + float edge = sgx_vbuf_env_f("SGX_HUD_EDGE", 60.0f); + float h = (float)v->ctx->fb.height, y; + + if (!count || !data || !v->vertex_size) + return 0; + y = ((const float *)data)[1]; + return y <= edge || y >= h - edge; +} + +/* Three attributes, always, whether the shader produces them or not: the + * vertex is a fixed record on this part, so a missing colour is a colour the + * hardware still reads. Emitting them unconditionally keeps the record's + * offsets constant, and the elements below simply omit the ones that have no + * source, which leaves sgx_draw()'s defaults in place. */ +/* Where the layout put an emitted attribute, in bytes into the draw module's + * vertex. The record's own offsets were literals - 16 for the colour, eight + * floats for the first set - which are only right while the emit order below + * never changes. Asking the layout is the same answer and cannot drift from + * it. */ +static unsigned sgx_vbuf_emit_off(const struct sgx_vbuf *v, unsigned k) +{ + static const unsigned char sz[] = { + [EMIT_1F] = 4, [EMIT_2F] = 8, [EMIT_3F] = 12, [EMIT_4F] = 16, + }; + unsigned i, off = 0; + + for (i = 0; i < k && i < v->vinfo.num_attribs; i++) { + unsigned e = v->vinfo.attrib[i].emit; + + off += e < sizeof sz / sizeof sz[0] && sz[e] ? sz[e] : 16u; + } + return off; +} + +static void sgx_vbuf_layout(struct sgx_vbuf *v) +{ + struct vertex_info *vi = &v->vinfo; + unsigned tex_mask = sgx_fs_tex_inputs(v->ctx); + const struct xpsb_attribs *at = sgx_fs_attribs(v->ctx); + int found[SGX_VBUF_MAX_VARY], nfound = 0, colour = -1, tex = -1; + unsigned si; + int fcset = sgx_fs_fragcoord_set(v->ctx); + int pos, i, sprite_uv; + + v->ctx->point_size_in_record = 0; + v->psize_dw = -1; + v->psize_from_vs = 0; + v->sprite_set = 0; + pos = draw_find_shader_output(v->draw, TGSI_SEMANTIC_POSITION, 0); + + /* The vertex output each fragment input reads, looked up by the + * semantic that input carries. + * + * This walked the semantics instead - colour, then generic, then + * texcoord, by semantic index - and called the result "the order a + * fragment program's inputs are declared". Nothing requires the two to + * agree: a program declaring a coordinate before a colour is numbered + * the other way round by that walk, and every later step indexes this + * by the input number, so the record carried one input's data where + * another's was read. Asking for each input's own semantic is the same + * answer wherever they do agree and the right one where they do not. */ + { + const unsigned char *fsem = NULL, *fidx = NULL; + unsigned nin = sgx_fs_inputs(v->ctx, &fsem, &fidx); + + for (i = 0; i < (int)nin && nfound < (int)SGX_VBUF_MAX_VARY; + i++) { + int at = draw_find_shader_output(v->draw, fsem[i], + fidx[i]); + + /* A fragment input the vertex program does not write + * still takes its place, so input n stays found[n]. */ + found[nfound++] = at; + } + } + + /* One of them is the colour the fragment program reads from pa0 and + * one is the coordinate the texture unit iterates. The program says + * which: a varying it samples with is a coordinate, and anything else + * is the colour. Getting this the wrong way round leaves the colour at + * its default and samples the texture at whatever the colour was. */ + for (i = 0; i < nfound; i++) { + if (tex_mask & (1u << i)) { + if (tex < 0) + tex = found[i]; + } else if (colour < 0) { + colour = found[i]; + } + } + if (colour < 0 && tex < 0 && nfound) + colour = found[0]; + /* The colour slot carries the input the fragment program was compiled + * to unpack, not simply the first one nothing samples. Mesa packs four + * of alacritty's varyings into three, and taking the first left the + * third never emitted at all: the record carried the second twice and + * the background colour lost the half that says how opaque it is. */ + { + int pi = sgx_fs_packed_in(v->ctx); + + if (pi >= 0 && pi < nfound) + colour = found[pi]; + } + + /* Both are off on the attribs path: a varying handed over as floats in + * a coordinate set must not also be emitted as the packed colour. It + * was, so the record carried the same value twice, the frame's + * iterator issued the packed dword into the first primary attribute, + * and a program compiled to read floats there got an 8888 dword + * reinterpreted as IEEE-754 - denormals near 1e-43, which write as + * zero. glmark2's gouraud rendered its whole canvas black for it. */ + /* A program with coordinate sets can still take one varying as the + * packed colour, and then the record's colour slot has to be written. + * Left at ctx_upload_vertices()' defaults it is four 1.0f, which the + * iterator packs to ffffffff - alacritty's background pass read that + * as an opaque blue and painted a rectangle behind every run of text + * where it should have discarded. */ + v->have_color = colour >= 0 && (!at || at->colour); + v->have_tex = tex >= 0 && !at; + v->attribs = at; + v->nfound = nfound; + if (sgx_vbuf_debug()) { + unsigned q; + + fprintf(sgx_log(), "sgx: vbuf: pos %d nfound %d colour %d tex %d " + "attribs %d have_color %d have_tex %d fs %p\n", pos, + nfound, colour, tex, at ? (int)at->nset : -1, + v->have_color, v->have_tex, + (const void *)sgx_ctx_fs(v->ctx)); + for (q = 0; at && q < at->nset; q++) + fprintf(sgx_log(), "sgx: vbuf: set %u width %u sampled %u" + " found %d\n", q, at->set[q].width, + at->set[q].sampled, + q < (unsigned)nfound ? found[q] : -1); + } + memcpy(v->found, found, sizeof found); + sgx_dbg("layout: %d varying(s), tex mask 0x%x -> colour %d, coord %d\n", + nfound, tex_mask, colour, tex); + + vi->num_attribs = 0; + draw_emit_vertex_attr(vi, EMIT_4F, pos < 0 ? 0 : pos); + /* A program that computes with a varying is handed it as floats in a + * coordinate set rather than as the packed colour, because a packed + * dword read as four floats saturates to nonsense. The record then + * carries it at its own width in the third slot, and the colour slot + * keeps whatever the shader put there. */ + draw_emit_vertex_attr(vi, EMIT_4F, colour < 0 ? 0 : colour); + /* SGX_POINT_SIZE_FIRST puts the size ahead of the coordinate sets + * instead of after them. The DDK's offsets ascend past the texcoords + * before reaching the size, so after is the reading; this is here to + * try the other one rather than argue about it. */ + if (sgx_vbuf_env_native_prim() && SGX_ENVS("SGX_POINT_SIZE_FIRST") && + draw_find_shader_output(v->draw, TGSI_SEMANTIC_PSIZE, 0) >= 0) { + /* psize_dw as well as the attribute: the emit copies the size + * into the record by that index, and without it the probe + * moved a dword nothing filled - which is how the offset + * sweeps came to answer nothing at all. */ + v->psize_dw = (int)(sgx_vbuf_emit_off(v, vi->num_attribs) / 4u); + v->psize_from_vs = 1; + draw_emit_vertex_attr(vi, EMIT_1F, + draw_find_shader_output(v->draw, + TGSI_SEMANTIC_PSIZE, 0)); + v->ctx->point_size_in_record = 1; + } + sprite_uv = sgx_sprite_uv_set(v->ctx); + v->sprite_set = sprite_uv; + if (sprite_uv) + draw_emit_vertex_attr(vi, EMIT_2F, pos < 0 ? 0 : pos); + if (at) { + /* One coordinate set per fragment input, in declaration order, + * each at the width the program's issue list claims. A shader + * that samples and shades needs both - the texel's coordinate + * and the iterated colour are different quantities and cannot + * share a set. */ + unsigned s; + + for (s = 0; s < at->nset; s++) { + /* Indexed by the set's width, so it has to cover + * every value the mask below can produce - it had + * five entries and was read with a three-bit index, + * which walked off the end for a width of five or + * more. */ + static const unsigned emit[8] = { + EMIT_3F, EMIT_3F, EMIT_2F, EMIT_3F, EMIT_4F, + EMIT_4F, EMIT_4F, EMIT_4F + }; + /* The varying this set was given, which is not the + * set's own position: sampled sets are numbered + * first so their index is a texissue the frame can + * hand a texture unit. */ + int fv = sgx_fs_set_varying(v->ctx, s); + unsigned fs = fv >= 0 ? (unsigned)fv : + (sgx_fs_sets_swapped(v->ctx) && + at->nset == 2 ? (s ^ 1u) : s); + int src = (int)fs < nfound ? found[fs] : -1; + + /* gl_FragCoord is fed from the position the epilogue + * already put in the record: interpolated across the + * triangle it is the window coordinate, which is what + * the fragment program is asking for. */ + if ((int)s == fcset) + src = pos; + if ((at->set[s].width & 7u) < 2u || + (at->set[s].width & 7u) > 4u) + fprintf(sgx_log(), "sgx: BAD ATTRIBS: set %u " + "width %u is not 2..4; the record and " + "the iterator will disagree about the " + "vertex\n", s, at->set[s].width); + draw_emit_vertex_attr(vi, emit[at->set[s].width & 7u], + src < 0 ? 0 : src); + } + } else if (!sprite_uv) { + draw_emit_vertex_attr(vi, EMIT_3F, tex < 0 ? 0 : tex); + } + /* A native point takes its size from the record. The ISP has no point + * width - the field at [31:28] sizes lines, and a sprite rasterises + * nothing whatever it is set to - so the size is a vertex output the + * MTE reads, the way the vendor emits FFGEN_OUTPUT_POINTSIZE. The + * offsets in the DDK put it after the coordinate sets and before the + * fog, so it goes on the end here, and bit 8 of the output selects + * tells the MTE to look for it. + * + * Only for a native point: a widened one is already the size it should + * be, and the extra dword would leave the record disagreeing with what + * the frame declares. + * + * The primitive is not known here - the draw module asks for the + * layout before it says what it is drawing - so this follows the + * vertex program instead: a program that writes a size gets a dword + * for it. A triangle ignores the size it is handed, and the record + * still says how wide it is, so carrying it costs one dword. */ + /* Only when the state says the program's size is the one to use: the + * draw module runs the whole program either way, so the output holds + * what the shader wrote whether or not GL_VERTEX_PROGRAM_POINT_SIZE is + * on, and reading it regardless drew glPointSize(3) nine wide. + * + * A native point needs the dword whether or not the program writes + * one: without it the frame cannot set the MTE's size bit and the + * sprite rasterises nothing at all. Where the program has no size the + * dword is a placeholder the emit fills with the rasterizer's + * constant, and psize_from_vs says which of the two it is. */ + if (!SGX_ENVS("SGX_POINT_SIZE_FIRST") && + (v->psize_per_vertex || sprite_uv)) { + /* SGX_NO_PT_PER_VERTEX asks for the rasterizer's constant + * rather than the program's output, which is a source for the + * dword and not a reason to leave it out: without the dword a + * native point has no size bit and draws nothing at all. */ + int psz = (v->psize_per_vertex && + !SGX_ENVS("SGX_NO_PT_PER_VERTEX")) ? + draw_find_shader_output(v->draw, + TGSI_SEMANTIC_PSIZE, 0) : -1; + + sgx_dbg("layout: point size output %d\n", psz); + if (psz >= 0 || sprite_uv) { + /* SGX_PT_SIZE_DW pads the record so the size lands at + * a chosen dword. The MTE reads it from somewhere the + * record has to match, and with a sprite that now + * rasterises the offset can be searched for rather + * than argued about. */ + const char *at_dw = SGX_ENVS("SGX_PT_SIZE_DW"); + unsigned want = at_dw && *at_dw ? + (unsigned)strtoul(at_dw, NULL, 0) : 0u; + + while (want && + sgx_vbuf_emit_off(v, vi->num_attribs) / 4u < + want) + draw_emit_vertex_attr(vi, EMIT_1F, + pos < 0 ? 0 : pos); + v->psize_dw = (int)(sgx_vbuf_emit_off(v, + vi->num_attribs) / 4u); + v->psize_from_vs = psz >= 0; + draw_emit_vertex_attr(vi, EMIT_1F, + psz >= 0 ? psz : + (pos < 0 ? 0 : pos)); + v->ctx->point_size_in_record = 1; + } + } + draw_compute_vertex_size(vi); + v->vertex_size = vi->size * 4; + /* What the module is asked to emit, and from which shader output. + * The JIT and the interpreter each build the post-vertex vertex + * from this, so a position read as -nan under one and not the + * other is answered here. */ + if (getenv("SGX_DUMP_VINFO")) { + unsigned q; + + fprintf(sgx_log(), "sgx: vinfo: %u attrib(s), size %u:", + vi->num_attribs, vi->size); + for (q = 0; q < vi->num_attribs; q++) + fprintf(sgx_log(), " [emit %u src %u]", + (unsigned)vi->attrib[q].emit, + (unsigned)vi->attrib[q].src_index); + fprintf(sgx_log(), "\n"); + } +} + +static const struct vertex_info *sgx_vbuf_get_vertex_info(struct vbuf_render *r) +{ + struct sgx_vbuf *v = (struct sgx_vbuf *)r; + + sgx_vbuf_layout(v); + return &v->vinfo; +} + +static bool sgx_vbuf_allocate_vertices(struct vbuf_render *r, + uint16_t vertex_size, uint16_t nr) +{ + struct sgx_vbuf *v = (struct sgx_vbuf *)r; + size_t want = (size_t)vertex_size * nr; + + if (want > v->verts_size) { + FREE(v->verts); + /* A guard past the end while SGX_VBUF_CANARY is set, so an + * overrun of this buffer is caught here rather than as the + * caller's heap failing an allocation much later. */ + v->verts = MALLOC(want + 64u); + v->verts_size = v->verts ? want : 0; + } + if (v->verts && sgx_vbuf_canary()) + memset((char *)v->verts + want, 0xa5, 64u); + v->vertex_size = vertex_size; + v->verts_used = want; + return v->verts != NULL; +} + +static void *sgx_vbuf_map_vertices(struct vbuf_render *r) +{ + return ((struct sgx_vbuf *)r)->verts; +} + +/* The coordinate the texture unit samples, in the set of its own it was given. + * + * The draw module emits a whole vertex output into a slot starting at its x, + * so a coordinate Mesa packed behind another varying arrives behind it here + * too. The sampled set is three floats projected - the third is the divisor - + * so that set is filled with the two components the coordinate is actually in + * and a divisor of one. The set behind it holds the same varying untouched, + * which is what the program's own reads are based on. */ +/* Every sampled set's divisor. + * + * A set the texture unit samples is three floats and the unit divides by the + * third - attrib_issues() asks for texdim 3, and asking for two stalls the + * render. A two-component texture coordinate only writes x and y, so that + * third float is whatever the vertex output register happened to hold: the + * coordinate came out divided by rubbish and the unit read past the end of the + * texture, which is an MMU fault with the cache as the requestor. A plain + * sample has no projection of its own - a projected one is TXP, which the + * iterated path refuses - so the divisor is one. */ +/* SGX_3D_S_BIAS, in slices: subtracted from every volume coordinate's + * slice component, scaled to the volume's depth. Zero unless set. It exists + * for one measurement - if the unit's slice centres turn out to sit at + * integers rather than at half-slices, 0.5 here is the correction - and it + * is the tex3dsweep case that says whether it is needed. */ +static float sgx_vbuf_s_bias(void) +{ + static int have = -1; + static float bias; + + if (have < 0) { + const char *e = getenv("SGX_3D_S_BIAS"); + + bias = e && *e ? strtof(e, NULL) : 0.0f; + have = 1; + } + return bias; +} + +static void sgx_vbuf_sampled_divisor(struct sgx_vbuf *v, uint16_t min, + uint16_t max) +{ + const struct xpsb_attribs *at = sgx_fs_attribs(v->ctx); + const struct sgx_shader *fs = sgx_ctx_fs(v->ctx); + int preiter = !fs || fs->tex_preiterated; + unsigned off = 8, s, i; + + static signed char nodiv = -1; + + if (!at || !v->verts || !v->vertex_size || + sgx_vbuf_env_set("SGX_NO_DIVISOR", &nodiv)) + return; + for (s = 0; s < at->nset; s++) { + unsigned w = at->set[s].width & 7u; + /* The third float, which is where a three-float sampled set's + * divisor belongs and where every set that renders correctly + * has had it. + * + * xpsb_attrib_proj_float() says T is the set's *last* float, + * so a four-float UVST set should divide by the fourth - and + * the record was moved there once. It is not that: a set + * divided by its fourth float would sample along a wrong but + * varying line, and both composite cases measure a mask that + * is the same at every pixel (work/fix-twosets/fixes2). The + * TAG is not dividing by that float at all, so the record + * keeps the placement every passing case uses until a probe + * says which float it really reads. SGX_PROJ_LAST follows the + * declared dimension, which is the experiment. */ + int pf = SGX_ENVS("SGX_PROJ_LAST") ? + xpsb_attrib_proj_float(at, s) : -1; + unsigned dv = pf >= 0 ? (unsigned)pf : 2u; + + if (at->set[s].projected && w >= 4 && + off + 4u <= v->vertex_size / 4u) { + /* A projected sample divided here rather than by the + * iterator: the coordinate reaches the texture unit + * already divided, so the set is an ordinary four + * float one and the sample is a plain one. The + * iterator interpolates perspective-correctly, so + * this is exact wherever the divisor is constant + * across the primitive - which is every coordinate + * GL defaults q to one for. */ + for (i = min; i <= max; i++) { + float *o = (float *)((char *)v->verts + + (size_t)i * + v->vertex_size) + off; + float q = o[3]; + + if (q != 0.0f && q != 1.0f) { + o[0] /= q; + o[1] /= q; + } + o[dv] = 1.0f; + } + } else if (at->set[s].volume && w >= 3 && + off + 3u <= v->vertex_size / 4u) { + /* The third float is the slice and stays the + * caller's, apart from the measurement bias. The + * fourth, where the set is four wide, is the 1.0 a + * projector would divide by - the record must carry + * a defined float there either way. */ + float bias = sgx_vbuf_s_bias(); + unsigned depth = sgx_fs_set_depth(v->ctx, s); + int four = w >= 4 && off + 4u <= v->vertex_size / 4u; + + for (i = min; i <= max; i++) { + float *o = (float *)((char *)v->verts + + (size_t)i * + v->vertex_size) + off; + + if (bias != 0.0f && depth) + o[2] -= bias / (float)depth; + if (four) + o[3] = 1.0f; + } + } else if (at->set[s].sampled && w >= 3 && preiter && + off + dv + 1u <= v->vertex_size / 4u) { + /* Only when the texture unit iterates the coordinate: + * it divides by the set's T and waits for it, so it + * is given one. A program that samples for itself + * takes the coordinate from the register, and a four + * wide set then carries two of them - Mesa packs two + * vec2 varyings into one vec4 - so writing a divisor + * over one of them would destroy the second + * coordinate. That is what left every two coordinate + * composite black on that path. */ + for (i = min; i <= max; i++) + *((float *)((char *)v->verts + + (size_t)i * v->vertex_size) + + off + dv) = 1.0f; + } + off += w; + } +} + +/* Move each set's coordinate to the front of the set. A set is sampled with + * its first two floats, and a coordinate Mesa packed behind another arrives + * at .zw - so the two sets that share one varying would otherwise sample with + * the same pair. */ +static void sgx_vbuf_pair_coords(struct sgx_vbuf *v, uint16_t min, + uint16_t max) +{ + const struct xpsb_attribs *at = sgx_fs_attribs(v->ctx); + unsigned off = 8, s, i; + + if (!at || !v->verts || !v->vertex_size) + return; + for (s = 0; s < at->nset; s++) { + unsigned w = at->set[s].width & 7u; + unsigned char c0 = 0, c1 = 1; + + if (sgx_vbuf_debug()) + fprintf(sgx_log(), "sgx: pair: set %u w %u coord %u,%u " + "(rc %d) off %u of %u\n", s, w, c0, c1, + sgx_fs_set_coord(v->ctx, s, &c0, &c1), off, + v->vertex_size / 4u); + if (sgx_fs_set_coord(v->ctx, s, &c0, &c1) || + (c0 == 0 && c1 == 1) || w < 4 || + off + w > v->vertex_size / 4u) { + off += w; + continue; + } + /* The coordinate only. sgx_vbuf_sampled_divisor() places the + * divisor, and it runs after this, which it has to: for the + * second set the divisor's float is one of the two this + * reads. */ + for (i = min; i <= max; i++) { + float *o = (float *)((char *)v->verts + + (size_t)i * v->vertex_size) + off; + float a = o[c0], b = o[c1]; + + o[0] = a; + o[1] = b; + } + off += w; + } +} + +static void sgx_vbuf_split_coord(struct sgx_vbuf *v, uint16_t min, + uint16_t max) +{ + unsigned char c[3]; + const struct xpsb_attribs *at = sgx_fs_attribs(v->ctx); + unsigned off = 8, s, i, w; + int split_set = -1; + int in = sgx_fs_coord_split(v->ctx, &split_set, c); + + if (in < 0 || split_set < 0 || !at || !v->verts || !v->vertex_size) + return; + if ((unsigned)split_set >= at->nset) + return; + for (s = 0; (int)s < split_set; s++) + off += at->set[s].width & 7u; + w = at->set[split_set].width & 7u; + if (w < 4 || off + w > v->vertex_size / 4u) + return; + for (i = min; i <= max; i++) { + float *o = (float *)((char *)v->verts + + (size_t)i * v->vertex_size) + off; + float a = o[c[0]], b = o[c[1]], k = o[c[2]]; + + o[0] = a; + o[1] = b; + o[2] = k; + o[3] = 1.0f; + } +} + +static void sgx_vbuf_unmap_vertices(struct vbuf_render *r, uint16_t min, + uint16_t max) +{ + /* The coordinate first. A set's divisor is written into its third + * float, which is one of the two a packed coordinate is read from - + * so moving the coordinate afterwards read a divisor of one where the + * coordinate's first component should have been, and the unit sampled + * one place for every pixel. */ + /* The module's own vertex, before anything here rewrites it: a + * position that is already -nan at this point came out of the + * vertex shader that way, and one that is not was made so here. */ + if (getenv("SGX_DUMP_DRAWVTX")) { + struct sgx_vbuf *dv = (struct sgx_vbuf *)r; + + if (dv->verts && dv->vertex_size >= 16u) { + const float *f = (const float *) + ((const char *)dv->verts + + (size_t)min * dv->vertex_size); + unsigned q, nf = dv->vertex_size / 4u; + + fprintf(sgx_log(), "sgx: drawvtx %u (stride %u):", + (unsigned)min, dv->vertex_size); + for (q = 0; q < nf && q < 12u; q++) + fprintf(sgx_log(), " %g", f[q]); + fprintf(sgx_log(), "\n"); + } + } + sgx_vbuf_pair_coords((struct sgx_vbuf *)r, min, max); + sgx_vbuf_sampled_divisor((struct sgx_vbuf *)r, min, max); + sgx_vbuf_split_coord((struct sgx_vbuf *)r, min, max); + { + struct sgx_vbuf *cv = (struct sgx_vbuf *)r; + + if (cv->verts && sgx_vbuf_canary()) { + const unsigned char *g = + (const unsigned char *)cv->verts + cv->verts_used; + unsigned i; + + for (i = 0; i < 64u; i++) + if (g[i] != 0xa5) { + fprintf(sgx_log(), "sgx: VERTEX BUFFER " + "OVERRUN at +%u (size %zu, " + "stride %u)\n", i, + cv->verts_used, + cv->vertex_size); + break; + } + } + } + (void)r; (void)min; (void)max; +} + +static void sgx_vbuf_set_primitive(struct vbuf_render *r, enum mesa_prim prim) +{ + struct sgx_vbuf *v = (struct sgx_vbuf *)r; + unsigned want; + + v->prim = prim; + + /* One frame rasterises one kind of object: the type and the width + * live in the ISP word the state block carries, so a change of + * primitive is a state change and the frame has to end. Points become + * sprites, which is the vendor's own mapping and what a later + * gl_PointCoord will need. */ + /* SGX_POINT_OBJTYPE varies which of the part's point shapes a point + * is sent as. The vendor's table says SPRITEUV, and a sprite may want + * a coordinate hole this record has no room for, so the two triangle + * forms - LINETRI and POINTTRI - are worth asking about too. */ + /* Points, yes; lines, not yet. A native line of any width costs a core + * recovery - `lineband` logs one with native primitives on and none + * with them off, and it reported a pass either way, so the recovery is + * what says it rather than the picture. Points cost none and fix three + * cases, so the two are separated rather than held together. + * SGX_NATIVE_LINE=1 hands lines to the ISP. */ + want = prim == MESA_PRIM_POINTS ? sgx_point_objtype(v) : + (sgx_vbuf_is_line(v) && + SGX_ENVS("SGX_NATIVE_LINE")) ? SGX_ISP_OBJ_LINE : + SGX_ISP_OBJ_TRI; + /* Widened on the CPU instead, and the object type has to say so here: + * the widening produces triangles, and a frame whose ISP word still + * named a sprite would rasterise each of the six corners as one. The + * scissor belongs with it - the clipper only cuts triangles, so a + * scissored point or line cannot go over as itself, and the test that + * refused it in the emit came too late to keep the object type + * honest. */ + if (!sgx_vbuf_env_native_prim() || sgx_vbuf_scissoring(v)) + want = SGX_ISP_OBJ_TRI; + /* One frame rasterises one shape, so taking the part's own point or + * line into a frame that already holds triangles ends the frame - and + * a submit is a tiler pass and a render that loads every tile back, + * which costs more than widening the handful of points that usually + * follow a triangle. So the native object is taken only where the + * frame is empty or already rasterising that shape; anywhere else the + * point or line is widened as it always was, and no frame ends that + * would not have ended anyway. A triangle cannot be widened into + * anything, so it still ends the frame. + * + * SGX_PRIM_SPLIT takes the native object whatever it costs, which is + * how the two are compared. */ + if (want != SGX_ISP_OBJ_TRI && v->ctx->draws && + v->ctx->prim_objtype == SGX_ISP_OBJ_TRI && + !SGX_ENVS("SGX_PRIM_SPLIT")) + want = SGX_ISP_OBJ_TRI; + if (want != v->ctx->prim_objtype) { + sgx_perf_note("primitive type"); + /* A record carries its own ISP object type and its own VDM + * command, so a frame can hold both shapes and the change + * only has to open another record. Ending the frame for it + * cost the X server two hundred submits painting one screen + * of text, where it alternates glyph points with fills. */ + if (v->ctx->draws && !sgx_mix_prim() && !sgx_flush(v->ctx)) + sgx_set_clear_enable(v->ctx, 0); + v->ctx->prim_objtype = want; + } +} + +static void sgx_vbuf_release_vertices(struct vbuf_render *r) +{ + (void)r; +} + +static void sgx_vbuf_destroy(struct vbuf_render *r) +{ + struct sgx_vbuf *v = (struct sgx_vbuf *)r; + + FREE(v->verts); + FREE(v->gather); + FREE(v->order); + free(v->clip); + free(v->clip_a); + free(v->clip_b); + free(v->line); + FREE(v); +} + + +/* Clip a triangle list to the scissor rectangle. + * + * Nothing in the reverse-engineered register set is a scissor, so it is done + * to the geometry instead: each triangle is cut against the four edges by + * Sutherland-Hodgman and the polygon that comes out is fanned back into + * triangles. The vertices reaching this backend are already in screen space - + * the draw module has transformed, divided and viewport-mapped them - so the + * rectangle is in the same units and clipping is a straight interpolation of + * every float in the record. + * + * Returns the number of vertices written to `out`, which the caller sizes at + * six triangles per input triangle: each of the four edges can add at most one + * vertex, so a triangle becomes at most a heptagon, which fans into five. + */ +#define SGX_CLIP_MAX_VERTS 8 + +static void clip_lerp(float *d, const float *a, const float *b, float t, + unsigned floats) +{ + unsigned i; + + for (i = 0; i < floats; i++) + d[i] = a[i] + (b[i] - a[i]) * t; +} + +/* Inside test and parameter for one edge. axis 0 is x, 1 is y; `keep_ge` says + * the inside is the side above the limit. */ +static float clip_dist(const float *v, int axis, float lim, int keep_ge) +{ + return keep_ge ? v[axis] - lim : lim - v[axis]; +} + +static unsigned clip_poly_edge(const float *in, unsigned n, float *out, + unsigned floats, int axis, float lim, + int keep_ge) +{ + unsigned i, m = 0; + + for (i = 0; i < n; i++) { + const float *a = in + (size_t)i * floats; + const float *b = in + (size_t)((i + 1) % n) * floats; + float da = clip_dist(a, axis, lim, keep_ge); + float db = clip_dist(b, axis, lim, keep_ge); + + if (da >= 0.0f) { + memcpy(out + (size_t)m * floats, a, floats * 4); + m++; + } + if ((da >= 0.0f) != (db >= 0.0f)) { + float t = da / (da - db); + + clip_lerp(out + (size_t)m * floats, a, b, t, floats); + m++; + } + if (m + 2 > SGX_CLIP_MAX_VERTS) + break; + } + return m; +} + +static unsigned sgx_vbuf_scissor_tris(struct sgx_vbuf *v, const float *in, + unsigned nverts, float *out) +{ + unsigned floats = v->vertex_size / 4; + const float sx0 = v->ctx->scissor_x0, sx1 = v->ctx->scissor_x1; + const float sy0 = v->ctx->scissor_y0, sy1 = v->ctx->scissor_y1; + float *a, *b; + unsigned t, n, k, m = 0; + + if (!floats) + return 0; + /* Grown to the record in use. It was a pair of fixed sixteen-float + * arrays with a bail above them, so a record wider than that - which + * is what a third coordinate set makes - had every one of its + * triangles dropped here, silently: ioquake3's world was clipped away + * in full, 1892 draws of a seventy-second run, and rendered flat. */ + if (floats > v->clip_ab_floats || !v->clip_a || !v->clip_b) { + float *na = realloc(v->clip_a, (size_t)SGX_CLIP_MAX_VERTS * + floats * sizeof *na); + float *nb = realloc(v->clip_b, (size_t)SGX_CLIP_MAX_VERTS * + floats * sizeof *nb); + + if (na) + v->clip_a = na; + if (nb) + v->clip_b = nb; + if (!na || !nb) { + fprintf(sgx_log(), "sgx: the clipper cannot size itself " + "for a %u float record; the draw is dropped\n", + floats); + return 0; + } + v->clip_ab_floats = floats; + } + a = v->clip_a; + b = v->clip_b; + for (t = 0; t + 2 < nverts; t += 3) { + const float *p0 = in + (size_t)t * floats; + const float *p1 = p0 + floats; + const float *p2 = p1 + floats; + float lo_x, hi_x, lo_y, hi_y; + + /* Trivially accepted: every vertex inside all four edges, by + * the clipper's own test (inside is clip_dist() >= 0) on the + * same floats and the same window space - x is [0], y is [1], + * whatever the winding. Each pass would then copy the polygon + * through untouched and the fan below would emit these three + * vertices in this order, so the memcpy is the same bytes. + * Read per vertex rather than off a box: a box built with + * min/max ignores a NaN coordinate, and this must not. Any + * comparison against NaN is false, so such a triangle falls + * through to the full clipper, which is where it was. */ + if (p0[0] >= sx0 && p0[0] <= sx1 && + p0[1] >= sy0 && p0[1] <= sy1 && + p1[0] >= sx0 && p1[0] <= sx1 && + p1[1] >= sy0 && p1[1] <= sy1 && + p2[0] >= sx0 && p2[0] <= sx1 && + p2[1] >= sy0 && p2[1] <= sy1) { + memcpy(out + (size_t)m * floats, p0, floats * 4 * 3); + m += 3; + continue; + } + /* Trivially rejected: the triangle's box wholly beyond one + * edge. That pass then finds no vertex inside and no crossing + * and returns nothing; every later pass only shrinks the + * polygon back into the box, so the box answers for all four. + * A NaN vertex is outside every edge here too, so leaving it + * out of the box is the same answer. */ + lo_x = hi_x = p0[0]; + if (p1[0] < lo_x) lo_x = p1[0]; + if (p1[0] > hi_x) hi_x = p1[0]; + if (p2[0] < lo_x) lo_x = p2[0]; + if (p2[0] > hi_x) hi_x = p2[0]; + lo_y = hi_y = p0[1]; + if (p1[1] < lo_y) lo_y = p1[1]; + if (p1[1] > hi_y) hi_y = p1[1]; + if (p2[1] < lo_y) lo_y = p2[1]; + if (p2[1] > hi_y) hi_y = p2[1]; + if (hi_x < sx0 || lo_x > sx1 || hi_y < sy0 || lo_y > sy1) + continue; + memcpy(a, p0, floats * 4 * 3); + n = 3; + n = clip_poly_edge(a, n, b, floats, 0, sx0, 1); + if (!n) continue; + n = clip_poly_edge(b, n, a, floats, 0, sx1, 0); + if (!n) continue; + n = clip_poly_edge(a, n, b, floats, 1, sy0, 1); + if (!n) continue; + n = clip_poly_edge(b, n, a, floats, 1, sy1, 0); + if (n < 3) continue; + /* Fan the polygon back into triangles. */ + for (k = 1; k + 1 < n; k++) { + memcpy(out + (size_t)m * floats, a, floats * 4); + m++; + memcpy(out + (size_t)m * floats, + a + (size_t)k * floats, floats * 4); + m++; + memcpy(out + (size_t)m * floats, + a + (size_t)(k + 1) * floats, floats * 4); + m++; + } + } + return m; +} + +static int sgx_vbuf_scissoring(const struct sgx_vbuf *v) +{ + /* Only a box that removes something. Clipping every triangle against + * the whole surface is work for no fragment - ioquake3 ran 4550 of + * them a frame. */ + return sgx_scissor_clips(v->ctx); +} + +/* Point the context at a run of hardware vertices and draw them. The elements + * name where each piece sits in the record the draw module just wrote; the + * ones with no source are left out, so the record's defaults stand. */ +static void sgx_vbuf_emit_raw(struct sgx_vbuf *v, const void *data, + unsigned count); + +/* Widen line segments into triangles. + * + * The draw module leaves width-one lines alone - draw_pipe_validate.c tests + * line_width != 1.0f - because it expects the part to rasterise them, and this + * one has no line rasteriser. Every point and line X draws is width one, so + * without this each one reached the backend as a primitive with no path and + * was dropped: PolyLine, PolySegment, PolyRectangle and the text cursor all + * disappeared. + * + * in holds 2 vertices per segment. Each becomes two triangles around the + * segment's own normal, the attributes taken from the endpoint each corner + * belongs to so anything interpolated along the line still is. */ +/* Where an axis-aligned line's band lies, in whole pixels: the first row (or + * column) the centre falls in, and one past the last. + * + * Extruded by half the width either side, a one-pixel line through y = 32.0 + * has its long edges at 31.5 and 32.5 - which is exactly where the rasteriser + * samples, so whether a row is covered is settled by the fill rule and by + * rounding rather than by the geometry. sgx_vbuf_widen_points() snaps a + * square for the same reason and records what leaving it unsnapped cost. A + * diagonal has no whole-pixel band and keeps the extrusion. */ +static void sgx_vbuf_line_band(float c, float hw, float *lo, float *hi) +{ + float w = hw * 2.0f < 1.0f ? 1.0f : floorf(hw * 2.0f + 0.5f); + + *lo = floorf(c - hw + 0.5f); + *hi = *lo + w; +} + +static unsigned sgx_vbuf_widen_lines(struct sgx_vbuf *v, const float *in, + unsigned nverts, float **out) +{ + unsigned floats = v->vertex_size / 4; + unsigned seg = nverts / 2, s, m = 0; + float hw = v->ctx->line_width * 0.5f; + /* SGX_NO_LINE_SNAP restores the unsnapped extrusion, to compare. */ + static signed char nosnap = -1; + int snap = !sgx_vbuf_env_set("SGX_NO_LINE_SNAP", &nosnap); + + if (!seg) + return 0; + if (seg * 6u > v->line_cap || v->vertex_size > v->line_stride) { + float *p = realloc(v->line, (size_t)seg * 6u * v->vertex_size); + + if (!p) + return 0; + v->line = p; + v->line_cap = seg * 6u; + v->line_stride = v->vertex_size; + } + if (hw < 0.5f) + hw = 0.5f; + for (s = 0; s < seg; s++) { + const float *a = in + (size_t)(2 * s) * floats; + const float *b = in + (size_t)(2 * s + 1) * floats; + float dx = b[0] - a[0], dy = b[1] - a[1]; + float len = sqrtf(dx * dx + dy * dy); + float nx, ny, lox, loy, hix, hiy; + unsigned k; + /* corner c takes its attributes from endpoint src[c] */ + static const unsigned char src[6] = { 0, 0, 1, 0, 1, 1 }; + static const signed char sgn[6] = { 1, -1, -1, 1, -1, 1 }; + + if (len < 1e-6f) { + nx = hw; ny = 0.0f; + } else { + nx = -dy / len * hw; ny = dx / len * hw; + } + lox = -nx; loy = -ny; hix = nx; hiy = ny; + /* Both endpoints share the coordinate the band is built on - + * a[1] for a flat segment, a[0] for an upright one - so one + * snap covers the whole quad. */ + if (snap && dy == 0.0f && dx != 0.0f) { + float lo, hi; + + sgx_vbuf_line_band(a[1], hw, &lo, &hi); + lox = hix = 0.0f; + loy = lo - a[1]; + hiy = hi - a[1]; + } else if (snap && dx == 0.0f && dy != 0.0f) { + float lo, hi; + + sgx_vbuf_line_band(a[0], hw, &lo, &hi); + loy = hiy = 0.0f; + lox = lo - a[0]; + hix = hi - a[0]; + } + for (k = 0; k < 6; k++) { + const float *e = src[k] ? b : a; + float *o = v->line + (size_t)m * floats; + + memcpy(o, e, v->vertex_size); + o[0] = e[0] + (sgn[k] > 0 ? hix : lox); + o[1] = e[1] + (sgn[k] > 0 ? hiy : loy); + m++; + } + } + *out = v->line; + return m; +} + +/* Whether coordinate set s is a point sprite's: gl_PointCoord's, or a + * TEXCOORD GL_COORD_REPLACE names. Found by the fragment input the set feeds, + * the same way sgx_vbuf_layout() laid the record out. */ +static int sgx_vbuf_set_is_sprite(const struct sgx_vbuf *v, unsigned s) +{ + const unsigned char *fsem = NULL, *fidx = NULL; + unsigned nin = sgx_fs_inputs(v->ctx, &fsem, &fidx); + int fv = sgx_fs_set_varying(v->ctx, s); + + if (fv < 0 || (unsigned)fv >= nin) + return 0; + return fsem[fv] == TGSI_SEMANTIC_PCOORD || + (fsem[fv] == TGSI_SEMANTIC_TEXCOORD && fidx[fv] < 32 && + ((v->sprite_enable >> fidx[fv]) & 1u)); +} + +/* The coordinate sets a point sprite writes, each with its offset in floats - + * the sets' widths summed behind the eight fixed dwords - and how many floats + * of it to write. Returns how many sets. */ +static unsigned sgx_vbuf_sprite_sets(const struct sgx_vbuf *v, + unsigned *off, unsigned char *wide) +{ + const struct xpsb_attribs *at = sgx_fs_attribs(v->ctx); + unsigned s, o = 8, n = 0; + + if (!at) + return 0; + for (s = 0; s < at->nset && s < XPSB_NSET_MAX; s++) { + unsigned w = at->set[s].width & 7u; + + if (sgx_vbuf_set_is_sprite(v, s) && + o + w <= v->vertex_size / 4u) { + off[n] = o; + /* A sampled set's third float is its divisor and is + * left alone; an iterated four-float one takes GL's + * (s, t, 0, 1). */ + wide[n] = (unsigned char)(at->set[s].sampled ? 2u : w); + n++; + } + o += w; + } + return n; +} + +/* SGX_PT_MIN raises a point's half-width, to tell a point that is too small + * for the tiler from one the tiler simply will not take. It is the knob that + * answers "does this path draw with points" of anything on screen - at six, + * a core-font glyph comes back as a block - so it has to mean the same thing + * whether the square is built here or the ISP rasterises the sprite. Read and + * converted once: this runs for every batch of points. */ +static float sgx_vbuf_pt_min(void) +{ + static float mn = -1.0f; + + if (mn < 0.0f) { + const char *e = SGX_ENVS("SGX_PT_MIN"); + + mn = e && *e ? (float)atof(e) : 0.0f; + if (mn < 0.0f) + mn = 0.0f; + } + return mn; +} + +/* Widen points into triangles, the way lines are widened just above and for + * the same reason: the part has no point rasteriser, and the position has + * already been through the viewport, so the square is built here in window + * coordinates. One vertex in, six out. + * + * A sprite's coordinate is written per corner: s left to right, t top to + * bottom in window space under PIPE_SPRITE_COORD_UPPER_LEFT and the other + * way under LOWER_LEFT - the mapping draw_pipe_wide_point.c set_texcoords() + * uses, in the same post-viewport space. Mesa's state tracker has already + * folded the framebuffer's orientation into the mode + * (st_atom_rasterizer.c). */ +static unsigned sgx_vbuf_widen_points(struct sgx_vbuf *v, const float *in, + unsigned nverts, float **out) +{ + unsigned floats = v->vertex_size / 4; + unsigned pt, m = 0; + unsigned soff[XPSB_NSET_MAX], nspr, q; + unsigned char swide[XPSB_NSET_MAX]; + /* The size the shader gave this point, if it gave one. gl_PointSize is + * per vertex and the rasterizer's constant is only the fallback: the + * draw module's own wide-point stage is refused here because it sizes + * against a viewport this driver's epilogue has already applied, so + * the per-vertex size has to be read out of the record instead. */ + float hw = v->ctx->point_size * 0.5f; + float mn = sgx_vbuf_pt_min(); + + if (!nverts) + return 0; + if (mn > hw) + hw = mn; + if (nverts * 6u > v->line_cap || v->vertex_size > v->line_stride) { + float *p = realloc(v->line, + (size_t)nverts * 6u * v->vertex_size); + + if (!p) + return 0; + v->line = p; + v->line_cap = nverts * 6u; + v->line_stride = v->vertex_size; + } + if (hw < 0.5f) + hw = 0.5f; + nspr = sgx_vbuf_sprite_sets(v, soff, swide); + for (pt = 0; pt < nverts; pt++) { + const float *a = in + (size_t)pt * floats; + /* The rasterizer's constant, not the record's dword. A + * per-vertex gl_PointSize is not available here: the draw + * module only computes that output when point_size_per_vertex + * is set, and setting it puts its own wide-point stage in the + * pipeline, which sizes against a viewport this driver's + * epilogue has already applied. Reading the slot regardless + * read whatever was left in it - the position, which drew a + * one pixel point sixteen wide. */ + float phw = hw; + + /* Unless the draw module computed one, in which case the slot + * holds the size the shader wrote. */ + if (v->psize_from_vs && v->psize_dw >= 0 && + (unsigned)v->psize_dw < floats) { + phw = a[v->psize_dw] * 0.5f; + if (mn > phw) + phw = mn; + if (phw < 0.5f) + phw = 0.5f; + } + /* two triangles over the square's corners */ + static const unsigned char cx[6] = { 0, 1, 1, 0, 1, 0 }; + static const unsigned char cy[6] = { 0, 0, 1, 0, 1, 1 }; + /* Snapped to the edges of the pixel the point falls in, so the + * square covers exactly that pixel and its corners are whole + * numbers. Built around the centre instead, every corner + * landed on a half coordinate - and the tiler stalled on it, + * which is what made a point never draw and left the X + * server's core-font text as filled blocks. */ + float x0 = floorf(a[0] - phw + 0.5f); + float y0 = floorf(a[1] - phw + 0.5f); + float w = phw * 2.0f < 1.0f ? 1.0f : + floorf(phw * 2.0f + 0.5f); + unsigned k; + + for (k = 0; k < 6; k++) { + float *o = v->line + (size_t)m * floats; + float t = v->sprite_upper_left ? (float)cy[k] : + 1.0f - (float)cy[k]; + + memcpy(o, a, v->vertex_size); + o[0] = x0 + w * (float)cx[k]; + o[1] = y0 + w * (float)cy[k]; + for (q = 0; q < nspr; q++) { + o[soff[q]] = (float)cx[k]; + o[soff[q] + 1] = t; + if (swide[q] >= 3) + o[soff[q] + 2] = 0.0f; + if (swide[q] >= 4) + o[soff[q] + 3] = 1.0f; + } + m++; + } + } + *out = v->line; + return m; +} + +/* Write the size a native point is to be drawn at into each vertex. + * + * The MTE reads the size out of the record, so the record has to hold one. + * The vertex program supplies it wherever it writes gl_PointSize and the + * layout emitted that output; where it does not, the dword is a placeholder + * and the size is the rasterizer's constant, which is what desktop GL and + * every core-font glyph the X server draws use. SGX_PT_MIN raises it here as + * it raises the widened square, so the knob means one thing on both paths. + * + * Returns the vertices to draw, which is the caller's own array when nothing + * had to be written, or 0 if the scratch could not be grown. + */ +/* SGX_PT_CENTRE=1 leaves the vertex at the point's centre, which is what the + * driver did before the corner was measured - the way back for a comparison. */ +static int sgx_vbuf_pt_corner_is_centre(void) +{ + static signed char on = -1; + + if (on < 0) + on = SGX_ENVS("SGX_PT_CENTRE") ? 1 : 0; + return on; +} + +static unsigned sgx_vbuf_point_sizes(struct sgx_vbuf *v, const float *in, + unsigned nverts, float **out) +{ + unsigned floats = v->vertex_size / 4; + float mn = sgx_vbuf_pt_min() * 2.0f; + float sz = v->ctx->point_size; + unsigned k; + + *out = NULL; + if (!nverts || v->psize_dw < 0 || (unsigned)v->psize_dw >= floats) + return nverts; + /* A size the vertex program wrote needs no clamp, but the vertex still + * has to be moved to the sprite's corner, so this no longer returns + * with the caller's own data - that early exit is what made the corner + * fix look like it did nothing. */ + if (v->psize_from_vs && !(mn > 0.0f) && sgx_vbuf_pt_corner_is_centre()) + return nverts; /* the record already holds it */ + if (sz < mn) + sz = mn; + if (sz < 1.0f) + sz = 1.0f; + if (nverts > v->line_cap || v->vertex_size > v->line_stride) { + float *p = realloc(v->line, (size_t)nverts * v->vertex_size); + + if (!p) + return 0; + v->line = p; + v->line_cap = nverts; + v->line_stride = v->vertex_size; + } + memcpy(v->line, in, (size_t)nverts * v->vertex_size); + for (k = 0; k < nverts; k++) { + float *o = v->line + (size_t)k * floats; + + if (v->psize_from_vs) { + if (o[v->psize_dw] < mn) + o[v->psize_dw] = mn; + } else { + o[v->psize_dw] = sz; + } + /* The part puts the sprite's own corner where the vertex is, + * not its centre: a point of size 8 at 16.5 lit 16..23 where + * 13..20 belongs, a constant three pixels on both axes and + * over three runs. So the vertex carries the corner, by the + * same expression the widening path already snaps to - which + * is what makes the two paths agree pixel for pixel rather + * than merely both being plausible. */ + if (!sgx_vbuf_pt_corner_is_centre()) { + float phw = o[v->psize_dw] * 0.5f; + + o[0] = o[0] - phw + 0.5f; + o[1] = o[1] - phw + 0.5f; + } + } + *out = v->line; + return nverts; +} + +void sgx_vbuf_set_sprite(struct vbuf_render *r, unsigned coord_enable, + int upper_left) +{ + struct sgx_vbuf *v = (struct sgx_vbuf *)r; + + v->sprite_enable = coord_enable; + v->sprite_upper_left = upper_left; +} + +void sgx_vbuf_set_psize_per_vertex(struct vbuf_render *r, int on) +{ + ((struct sgx_vbuf *)r)->psize_per_vertex = on; +} + +/* Emit, clipping to the scissor rectangle first when one is in force. The + * clip can turn one triangle into five, so the scratch is sized for that. */ +static int sgx_vbuf_is_line(const struct sgx_vbuf *v) +{ + return v->prim == MESA_PRIM_LINES || v->prim == MESA_PRIM_LINE_STRIP || + v->prim == MESA_PRIM_LINE_LOOP; +} + +/* Whether a point or a line goes to the ISP as itself. + * + * Points become SPRITEUV, lines LINE - the mapping the vendor's own primitive + * table uses (opengles2/validate.c:59-66). Strips and loops are already + * expanded to lists before here, which is required anyway: the line-strip VDM + * type is a later core's (sgxdefs.h:975, inside #if defined(SGX545)). + * + * On by default. It was behind the knob while a native point rasterised + * nothing, which was the record carrying no size for the MTE to read; with + * the size in the record a point draws at the size it was asked for. The part + * has no erratum against either object on this core - the two the DDK carries + * for them, BRN 29546 for points and BRN 31728 for lines, are both inside the + * SGX545 block of sgxerrata.h (:2592, :2650) and absent from the SGX535 one - + * and the vendor draws every point and line this way, with no software + * fallback anywhere in its stack. + * + * SGX_NATIVE_PRIM=0 widens every point and line into triangles on the CPU + * again, which is the whole of the way back. */ +static int sgx_vbuf_env_native_prim(void) +{ + static signed char on = -1; + + if (on < 0) { + const char *e = SGX_ENVS("SGX_NATIVE_PRIM"); + + on = !(e && *e == '0'); + } + return on; +} + +/* Settled in sgx_vbuf_set_primitive(), which is where the object type the + * frame carries is chosen: the two have to agree, or the ISP rasterises one + * shape from geometry built as another. */ +static int sgx_vbuf_native_prim(const struct sgx_vbuf *v) +{ + return v->ctx->prim_objtype != SGX_ISP_OBJ_TRI; +} + +static void sgx_vbuf_emit(struct sgx_vbuf *v, const void *data, unsigned count) +{ + static signed char dump = -1; + unsigned n; + + if (sgx_vbuf_debug()) { + const struct xpsb_attribs *dat = sgx_fs_attribs(v->ctx); + + fprintf(sgx_log(), "sgx: vbuf: emit %u vert(s), prim %u, fs %p, " + "%d set(s)\n", count, (unsigned)v->prim, + (const void *)sgx_ctx_fs(v->ctx), + dat ? (int)dat->nset : -1); + } + /* The record as the iterator will read it. A coordinate that never + * varies, or one that is not a number, is a texture fetch at an + * address nothing maps. */ + if (sgx_vbuf_env_set("SGX_DUMP_VERTS", &dump) && data && v->vertex_size) { + unsigned q, e, nf = v->vertex_size / 4u; + + for (q = 0; q < count && q < 3; q++) { + const float *f = (const float *)((const char *)data + + (size_t)q * v->vertex_size); + + fprintf(sgx_log(), "sgx: vert %u:", q); + for (e = 0; e < nf; e++) + fprintf(sgx_log(), " %g", f[e]); + fprintf(sgx_log(), "\n"); + } + } + + if (sgx_vbuf_hud_trace() && sgx_vbuf_in_band(v, data, count)) + fprintf(sgx_log(), "sgx: hud: emit %u vtx at %.1f,%.1f prim %u " + "scissor %d %g,%g..%g,%g\n", count, + ((const float *)data)[0], ((const float *)data)[1], + (unsigned)v->prim, sgx_vbuf_scissoring(v), + v->ctx->scissor_x0, v->ctx->scissor_y0, + v->ctx->scissor_x1, v->ctx->scissor_y1); + + /* Widened into triangles unless the ISP is rasterising the primitive + * itself, in which case the vertices go over as they are and the VDM + * takes one or two indices at a time. The scissor clipper below only + * knows about triangles, so a native point or line is emitted raw and + * the frame keeps a scissor it cannot apply - which is why the native + * path is refused while one is in force, in sgx_vbuf_native_prim(). */ + if (sgx_vbuf_native_prim(v)) { + /* A native point's size lives in the record, and the vertex + * program does not always write one. */ + if (v->prim == MESA_PRIM_POINTS && count) { + float *sized = NULL; + + if (!sgx_vbuf_point_sizes(v, data, count, &sized)) { + sgx_dbg("draw: no room to size %u native " + "point(s), so the draw is lost\n", + count); + return; + } + if (sized) + data = sized; + } + sgx_vbuf_emit_raw(v, data, count); + return; + } + if (sgx_vbuf_is_line(v) && count) { + float *wide = NULL; + + count = sgx_vbuf_widen_lines(v, data, count, &wide); + if (!count) + return; + if (wide) + data = wide; + } else if (v->prim == MESA_PRIM_POINTS && count) { + float *wide = NULL; + + count = sgx_vbuf_widen_points(v, data, count, &wide); + if (!count) + return; + data = wide; + } + if (!sgx_vbuf_scissoring(v) || !count) { + sgx_vbuf_emit_raw(v, data, count); + return; + } + /* The stride as well as the count: a record is as wide as the bound + * program needs - eleven floats, or twelve with an iterated varying, + * or more - so a buffer big enough in vertices can still be too small + * in bytes. Growing on the count alone let a wider record write past + * the end of it, which corrupted the caller's heap and aborted the + * process a few scenes into glmark2. */ + if (count / 3u > v->clip_cap / 15u || v->vertex_size > v->clip_stride) { + unsigned want = (count / 3u + 1u) * 15u; + float *p; + + if (want < v->clip_cap) + want = v->clip_cap; + p = realloc(v->clip, (size_t)want * v->vertex_size); + if (!p) { + sgx_vbuf_emit_raw(v, data, count); + return; + } + v->clip = p; + v->clip_cap = want; + v->clip_stride = v->vertex_size; + } + n = sgx_vbuf_scissor_tris(v, data, count, v->clip); + sgx_dbg("scissor: %u vertices in, %u out\n", count, n); + if (n) + sgx_vbuf_emit_raw(v, v->clip, n); +} + +static void sgx_vbuf_emit_raw(struct sgx_vbuf *v, const void *data, + unsigned count) +{ + /* Sized by what the context accepts, not by what the record used to + * need: position, four colour components and one coordinate set was + * exactly six, and a second iterated set made it seven - which this + * wrote past the end of, corrupting the stack. */ + struct sgx_vertex_element ve[SGX_MAX_VERTEX_ELEMENTS]; + struct sgx_vertex_buffer vb; + unsigned n = 0; + + /* every field of every element, rhw included: they are set per use */ + memset(ve, 0, sizeof ve); + + if (sgx_vbuf_debug() && count) { + const float *f = data; + unsigned k; + + fprintf(sgx_log(), "sgx: verts %u x %u bytes (psize dw %d):", + count, v->vertex_size, v->psize_dw); + { + unsigned d; + + fprintf(sgx_log(), " raw["); + for (d = 0; d < v->vertex_size / 4u; d++) + fprintf(sgx_log(), "%s%.3f", d ? " " : "", + f[d]); + fprintf(sgx_log(), "]"); + } + for (k = 0; k < count && k < 3; k++) { + const float *p = f + k * v->vertex_size / 4; + + fprintf(sgx_log(), " pos(%.2f %.2f %.5f %.5f)", + p[0], p[1], p[2], p[3]); + if (v->have_color) + fprintf(sgx_log(), " col(%.3f %.3f %.3f %.3f)", + p[4], p[5], p[6], p[7]); + /* Every coordinate set as the record carries it, so a + * sampler that reads one texel can be told apart from + * a record that carries one coordinate. */ + { + const struct xpsb_attribs *at = + sgx_fs_attribs(v->ctx); + unsigned q, off = 8; + + for (q = 0; at && q < at->nset; q++) { + unsigned w = at->set[q].width & 7u; + + if (off + w > v->vertex_size / 4u) + break; + fprintf(sgx_log(), " set%u[%s%u](", q, + at->set[q].sampled ? "s" : "", + w); + for (unsigned e = 0; e < w; e++) + fprintf(sgx_log(), "%s%.3f", + e ? " " : "", + p[off + e]); + fputc(')', stderr); + off += w; + } + } + } + fputc('\n', stderr); + } + vb.data = data; + vb.offset = 0; + vb.stride = v->vertex_size; + vb.size = (uint64_t)v->vertex_size * count; + + /* Three components copied, and the fourth handled by the rhw flag: the + * draw module leaves 1/w there, which is exactly the RHW plane the + * record's fourth float is - the value every iteration is perspective + * corrected against. The upload copies it through when it is sane and + * leaves the affine 1.0 otherwise. The old note here recorded 1/w as + * collapsing triangles to slivers; that measurement did not + * reproduce - with the same frame state the quad stays where it is + * and only the iteration changes, which is the DDK's account of the + * field (WPRESENT in the output selects). sgxtri's gears write 1.0 + * because they are affine by construction. */ + ve[n].src_offset = 0; ve[n].vb_index = 0; ve[n].ncomp = 3; + ve[n].stride = v->vertex_size; ve[n].slot = 0; + ve[n].rhw = 1; n++; + if (v->have_color) { + /* Blue first. The frame's iterator packs the record's colour + * floats into one 8888 dword and the first of them becomes the + * low byte, which an A8R8G8B8 target reads as blue - so handing + * it R,G,B,A put every gear's red into the blue channel. + * Measured: 0.758 arrived as 0xc1 in byte 0, the right value in + * the wrong place. Four single-component elements rather than + * one of four, because the order is what is being changed. */ + /* The colour's own bytes, found in the layout rather than + * written out: these were the literal offsets of the second + * emitted attribute, which is only where the colour is while + * nothing changes the order it is emitted in. ord[] is the + * permutation, z y x w, and the base is where the layout put + * it. */ + static const unsigned char ord[4] = { 8, 4, 0, 12 }; + unsigned base = sgx_vbuf_emit_off(v, 1); + unsigned c; + + for (c = 0; c < 4; c++) { + ve[n].src_offset = base + ord[c]; ve[n].vb_index = 0; + ve[n].ncomp = 1; ve[n].stride = v->vertex_size; + ve[n].slot = 4 + c; n++; + } + } + if (v->attribs) { + /* Each set at its own width, laid out behind the colour in + * declaration order. A sampled set keeps the old rule that + * only two of its three floats are written, because the third + * is the 1.0 the record already holds. */ + /* Behind the ISP's own coordinate set when the frame carries + * one: sgx_context.c puts the sprite's UV ahead of the + * program's sets, and the record has to be read the same way + * or every set lands two dwords early. */ + unsigned s, off = v->sprite_set ? 10u : 8u; + /* The draw module lays its vertex out in the record's own + * unrouted shape, so a set is read where it would have been + * and written where the frame now expects it: behind both + * colour quads when one of them carries a set. */ + unsigned src = off; + + if (!v->sprite_set) + off = xpsb_attrib_set_base(v->attribs); + + for (s = 0; s < v->attribs->nset && n < SGX_MAX_VERTEX_ELEMENTS; + s++) { + unsigned w = v->attribs->set[s].width; + + /* A sprite's set has no vertex output behind it; + * the widening wrote it and it has to go over. */ + if (((int)s < v->nfound && v->found[s] >= 0) || + (int)s == sgx_fs_fragcoord_set(v->ctx) || + (v->prim == MESA_PRIM_POINTS && + sgx_vbuf_set_is_sprite(v, s))) { + unsigned at = v->attribs->set[s].on_colour ? + (v->attribs->set[s].on_colour == 2u ? + 8u : 4u) : off; + + /* A set carried on a colour iterator goes into + * a colour quad, and the iterator packs that + * quad z y x w - the same permutation the + * packed colour above is written in. Four + * single-component elements, because the order + * is what is being changed; the components a + * narrower set does not have land in registers + * it never reads. */ + if (v->attribs->set[s].on_colour) { + static const unsigned char ord[4] = { + 8, 4, 0, 12 + }; + unsigned c; + + for (c = 0; c < 4 && + n < SGX_MAX_VERTEX_ELEMENTS; c++) { + ve[n].src_offset = + src * 4 + ord[c]; + ve[n].vb_index = 0; + ve[n].ncomp = 1; + ve[n].stride = v->vertex_size; + ve[n].slot = + (unsigned char)(at + c); + n++; + } + src += w; + continue; + } + ve[n].src_offset = src * 4; + ve[n].vb_index = 0; + /* A sampled set's third float is the 1.0 the + * record already holds - but a volume's + * coordinate set is iterated, not sampled, + * and every float of it is coordinate. */ + ve[n].ncomp = (unsigned char) + (v->attribs->set[s].sampled && + !v->attribs->set[s].volume ? 2 : w); + ve[n].stride = v->vertex_size; + ve[n].slot = (unsigned char)at; + n++; + } + src += w; + if (!v->attribs->set[s].on_colour) + off += w; + } + } + /* The point size, which nothing here used to copy. The layout appends + * the dword and the frame declares it - group 10's size bit and the + * record's width both follow it - but no element named it, so + * ctx_upload_vertices() left the slot at the 1.0f it initialises the + * record with and every gl_PointSize drew one pixel. Its record dword + * is behind the sets, which is where the layout emitted it. */ + if (v->psize_dw >= 0 && v->ctx->point_size_in_record && + n < SGX_MAX_VERTEX_ELEMENTS) { + /* Source dword and record dword are the same one: the layout + * above builds the draw module's vertex in the record's own + * shape, which is what lets the sets be copied at their own + * offsets. A record too narrow for it is refused by + * sgx_set_vertex_elements() rather than written short. */ + ve[n].src_offset = (unsigned)v->psize_dw * 4u; + ve[n].vb_index = 0; + ve[n].ncomp = 1; + ve[n].stride = v->vertex_size; + ve[n].slot = (unsigned char)v->psize_dw; + n++; + } + if (v->have_tex) { + /* Two components, not three: the record's third coordinate + * float is the 1.0 xpsb_gen_quad_n() writes, and a + * two-component varying's third channel is zero. Writing that + * zero over it made every fragment sample the same texel. */ + ve[n].src_offset = 32; ve[n].vb_index = 0; ve[n].ncomp = 2; + ve[n].stride = v->vertex_size; ve[n].slot = 8; n++; + } + /* The record has to be as wide as the elements just laid out. It is + * sized when a program is bound, but the layout is built here from the + * same attributes, and anything rebound in between left it at the + * eleven floats of a program with no varyings - so a second coordinate + * set ran off the end and the whole draw was refused. That is every + * masked composite the X server makes. */ + sgx_update_vtx_floats(v->ctx); + { + int rb = sgx_set_vertex_buffers(v->ctx, &vb, 1); + int re = rb ? 0 : sgx_set_vertex_elements(v->ctx, ve, n); + + if (getenv("SGX_DUMP_ELEMENTS")) { + unsigned q; + + for (q = 0; q < n; q++) + fprintf(sgx_log(), "sgx: element %u: slot %u " + "ncomp %u offset %u\n", q, + (unsigned)ve[q].slot, + (unsigned)ve[q].ncomp, + (unsigned)ve[q].src_offset); + } + + if (rb || re) { + unsigned q; + + if (sgx_vbuf_hud_trace()) + fprintf(sgx_log(), "sgx: hud: the vertex layout " + "was refused: buffers %d, elements %d, " + "%u element(s), record %u float(s)\n", + rb, re, n, v->ctx->vtx_floats); + sgx_dbg("draw: the vertex layout was refused: " + "buffers %d, elements %d, %u element(s), " + "record %u float(s)\n", rb, re, n, + v->ctx->vtx_floats); + for (q = 0; q < n; q++) + sgx_dbg("draw: element %u: slot %u ncomp %u " + "offset %u\n", q, + (unsigned)ve[q].slot, + (unsigned)ve[q].ncomp, + (unsigned)ve[q].src_offset); + return; + } + } + { + int ret = sgx_draw(v->ctx, count); + const float *p0 = (const float *)data; + + if (sgx_vbuf_hud_trace() && sgx_vbuf_in_band(v, data, count)) { + unsigned q, fl = v->vertex_size / 4u; + + fprintf(sgx_log(), "sgx: hud: draw %u vtx at %.1f,%.1f " + "vsz %u nset %d colour %d hc %d ret %d " + "records %u vtx_floats %u | v0", + count, p0[0], p0[1], v->vertex_size, + v->attribs ? (int)v->attribs->nset : -1, + v->attribs ? (int)v->attribs->colour : -1, + v->have_color, ret, v->ctx->nrange, + v->ctx->vtx_floats); + for (q = 0; q < fl; q++) + fprintf(sgx_log(), " %g", p0[q]); + fputc('\n', stderr); + } + + /* One frame holds a bounded number of vertices, and a model + * that outgrows it used to lose every draw past the boundary. + * Submit what has accumulated and carry on into the same + * target instead: the continuation loads each tile back rather + * than clearing, so the earlier draws survive it. */ + if (ret == -ENOSPC && !v->ctx->draws) + sgx_dbg("draw: no room and nothing to flush, so the " + "draw is lost\n"); + if (ret == -ENOSPC && v->ctx->draws) { + int fr = sgx_flush(v->ctx); + + if (fr) + sgx_dbg("draw: the flush that would make room " + "failed with %d; the draw is lost\n", + fr); + if (!fr) { + sgx_set_clear_enable(v->ctx, 0); + /* The flush sized the record for the frame it + * just submitted, which is the program that + * was bound then rather than the one this + * draw is waiting to use. Size it again for + * the bound program, or the elements below + * are refused for naming floats past the end + * of a record that is no longer this wide - + * and then the draw is never retried at all, + * ret keeps the -ENOSPC it came in with, and + * the whole frame is marked invalid. */ + sgx_update_vtx_floats(v->ctx); + if (!sgx_set_vertex_buffers(v->ctx, &vb, 1) && + !sgx_set_vertex_elements(v->ctx, ve, n)) + ret = sgx_draw(v->ctx, count); + sgx_dbg("draw: split the frame, %d\n", ret); + } + } + sgx_dbg("draw: %u vertices, %u element(s), %d%s%s%s\n", count, n, + ret, v->ctx->vs ? "" : " (no vertex program)", + v->ctx->fs ? "" : " (no fragment program)", + v->ctx->cookie_valid ? "" : " (no framebuffer)"); + /* A refused draw used to be reported to the debug log and + * nowhere else, so the frame went to the hardware missing the + * geometry it was told about. Mark the frame instead: the + * flush refuses it rather than submitting something that does + * not describe what was drawn. */ + if (ret) + v->ctx->frame_invalid = 1; + } +} + +/* The frame draws one triangle list, so a strip, a fan or a quad is expanded + * here. Points and lines have no path on this part yet and are dropped rather + * than drawn as something else. */ +static unsigned sgx_vbuf_tris(struct sgx_vbuf *v, unsigned nr, unsigned *out) +{ + /* The MTE's flat-shade field names the corner the colour is taken + * from - the last, or the first under flatshade_first - so GL's + * provoking vertex has to sit at that corner; see sgx_hwtcl_corner() + * for the same rule on the part's transform path. */ + unsigned first = v->ctx->rast.flatshade_first; + unsigned flat = v->ctx->rast.flatshade; + unsigned i, n = 0; + + switch (v->prim) { + case MESA_PRIM_TRIANGLES: + for (i = 0; i + 2 < nr; i += 3) { + out[n++] = i; out[n++] = i + 1; out[n++] = i + 2; + } + break; + case MESA_PRIM_TRIANGLE_STRIP: + for (i = 0; i + 2 < nr; i++) { + if (!(i & 1)) { + out[n++] = i; out[n++] = i + 1; out[n++] = i + 2; + } else if (first) { + out[n++] = i; out[n++] = i + 2; out[n++] = i + 1; + } else { + out[n++] = i + 1; out[n++] = i; out[n++] = i + 2; + } + } + break; + case MESA_PRIM_TRIANGLE_FAN: + for (i = 1; i + 1 < nr; i++) { + if (first) { + out[n++] = i; out[n++] = i + 1; out[n++] = 0; + } else { + out[n++] = 0; out[n++] = i; out[n++] = i + 1; + } + } + break; + /* A polygon's provoking vertex is its first under either convention. */ + case MESA_PRIM_POLYGON: + for (i = 1; i + 1 < nr; i++) { + if (first) { + out[n++] = 0; out[n++] = i; out[n++] = i + 1; + } else { + out[n++] = i; out[n++] = i + 1; out[n++] = 0; + } + } + break; + /* Quads reach here whole: the draw module hands the backend the + * caller's own mode when no pipeline stage runs, and dropping them + * lost every gear tooth in glxgears. A quad's provoking vertex is + * its fourth, which only the other diagonal puts in both triangles - + * taken for a flat-shaded draw alone. */ + case MESA_PRIM_QUADS: + for (i = 0; i + 3 < nr; i += 4) { + if (flat && !first) { + out[n++] = i; out[n++] = i + 1; out[n++] = i + 3; + out[n++] = i + 1; out[n++] = i + 2; out[n++] = i + 3; + } else { + out[n++] = i; out[n++] = i + 1; out[n++] = i + 2; + out[n++] = i; out[n++] = i + 2; out[n++] = i + 3; + } + } + break; + case MESA_PRIM_QUAD_STRIP: + for (i = 0; i + 3 < nr; i += 2) { + out[n++] = i; out[n++] = i + 1; out[n++] = i + 3; + if (first) { + out[n++] = i; out[n++] = i + 3; out[n++] = i + 2; + } else { + out[n++] = i + 2; out[n++] = i; out[n++] = i + 3; + } + } + break; + /* One index per point; sgx_vbuf_emit() widens it to two triangles. */ + case MESA_PRIM_POINTS: + for (i = 0; i < nr; i++) + out[n++] = i; + break; + /* Two indices per segment; sgx_vbuf_emit() widens them to triangles. */ + case MESA_PRIM_LINES: + for (i = 0; i + 1 < nr; i += 2) { + out[n++] = i; out[n++] = i + 1; + } + break; + case MESA_PRIM_LINE_STRIP: + for (i = 0; i + 1 < nr; i++) { + out[n++] = i; out[n++] = i + 1; + } + break; + case MESA_PRIM_LINE_LOOP: + for (i = 0; i + 1 < nr; i++) { + out[n++] = i; out[n++] = i + 1; + } + if (nr > 2) { out[n++] = nr - 1; out[n++] = 0; } + break; + default: + /* Counted, not only logged: the flush reported "0 dropped" + * while discarding every point and line glamor asked for. */ + v->ctx->prim_dropped++; + sgx_dbg("draw: primitive %u has no path and was dropped\n", + v->prim); + break; + } + return n; +} + +/* Sized from the record the draw module actually hands over, not from a fixed + * width. It was a hard-coded 44 bytes a vertex, which is what an untextured + * record measures; a program handed a varying as a coordinate set gets 48, and + * the gather then wrote four bytes past the buffer for every vertex it + * expanded. That was the heap corruption glmark2's texture scene aborted on. */ +static float *sgx_vbuf_gather_room(struct sgx_vbuf *v, unsigned verts) +{ + size_t want = (size_t)verts * v->vertex_size; + + if (want > v->gather_size) { + FREE(v->gather); + v->gather = MALLOC(want); + v->gather_size = v->gather ? want : 0; + } + return v->gather; +} + +/* The same grown-on-demand pattern for the expanded index order. It was a + * MALLOC and a FREE per draw, which is per primitive batch and shows up in a + * profile of anything that draws strips. */ +static unsigned *sgx_vbuf_order_room(struct sgx_vbuf *v, unsigned nr) +{ + size_t want = (size_t)nr * 3u + 3u; + + if (want > v->order_cap) { + FREE(v->order); + v->order = MALLOC(want * sizeof *v->order); + v->order_cap = v->order ? want : 0; + } + return v->order; +} + +static void sgx_vbuf_draw_elements(struct vbuf_render *r, + const uint16_t *indices, unsigned nr) +{ + struct sgx_vbuf *v = (struct sgx_vbuf *)r; + unsigned *order, n, i; + float *g; + + /* A triangle list expands into the identity, so the order pass and the + * indirection through it are both skipped: the gather reads the + * caller's indices straight, which is what draw_arrays() already does + * for the same primitive. */ + if (v->prim == MESA_PRIM_TRIANGLES) { + order = NULL; + n = nr - nr % 3; + } else { + order = sgx_vbuf_order_room(v, nr); + if (!order) + return; + n = sgx_vbuf_tris(v, nr, order); + } + g = n ? sgx_vbuf_gather_room(v, n) : NULL; + if (g) { + const char *src = (const char *)v->verts; + unsigned stride = v->vertex_size; + + if (order) + for (i = 0; i < n; i++) + memcpy((char *)g + (size_t)i * stride, + src + (size_t)indices[order[i]] * stride, + stride); + else + for (i = 0; i < n; i++) + memcpy((char *)g + (size_t)i * stride, + src + (size_t)indices[i] * stride, + stride); + sgx_vbuf_emit(v, g, n); + } +} + +static void sgx_vbuf_draw_arrays(struct vbuf_render *r, unsigned start, + unsigned nr) +{ + struct sgx_vbuf *v = (struct sgx_vbuf *)r; + const char *base = (const char *)v->verts + + (size_t)start * v->vertex_size; + unsigned *order, n, i; + float *g; + + if (v->prim == MESA_PRIM_TRIANGLES) { + sgx_vbuf_emit(v, base, nr - nr % 3); + return; + } + order = sgx_vbuf_order_room(v, nr); + if (!order) + return; + n = sgx_vbuf_tris(v, nr, order); + g = n ? sgx_vbuf_gather_room(v, n) : NULL; + if (g) { + unsigned stride = v->vertex_size; + + for (i = 0; i < n; i++) + memcpy((char *)g + (size_t)i * stride, + base + (size_t)order[i] * stride, stride); + sgx_vbuf_emit(v, g, n); + } +} + +struct vbuf_render *sgx_vbuf_create(struct sgx_context *ctx, + struct draw_context *draw) +{ + struct sgx_vbuf *v = CALLOC_STRUCT(sgx_vbuf); + + if (!v) + return NULL; + v->ctx = ctx; + v->draw = draw; + v->prim = MESA_PRIM_TRIANGLES; + v->sprite_upper_left = 1; + v->psize_per_vertex = 1; + /* The vertex buffer holds one frame's worth; the frame's own limit is + * what actually bounds a draw, and sgx_draw() reports going past it. */ + v->base.max_indices = 16384; + v->base.max_vertex_buffer_bytes = 1024 * 1024; + v->base.get_vertex_info = sgx_vbuf_get_vertex_info; + v->base.allocate_vertices = sgx_vbuf_allocate_vertices; + v->base.map_vertices = sgx_vbuf_map_vertices; + v->base.unmap_vertices = sgx_vbuf_unmap_vertices; + v->base.set_primitive = sgx_vbuf_set_primitive; + v->base.draw_elements = sgx_vbuf_draw_elements; + v->base.draw_arrays = sgx_vbuf_draw_arrays; + v->base.release_vertices = sgx_vbuf_release_vertices; + v->base.destroy = sgx_vbuf_destroy; + return &v->base; +} diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_pipe_vbuf.h mesa-26.2.2/src/gallium/drivers/sgx/sgx_pipe_vbuf.h --- mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_pipe_vbuf.h 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/sgx_pipe_vbuf.h 2026-09-08 10:57:36.680339645 +0200 @@ -0,0 +1,27 @@ +/* The draw module's rasterisation backend for this driver. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#ifndef _SGX_PIPE_VBUF_H_ +#define _SGX_PIPE_VBUF_H_ + +struct sgx_context; +struct draw_context; +struct vbuf_render; + +struct vbuf_render *sgx_vbuf_create(struct sgx_context *ctx, + struct draw_context *draw); + +/* Point sprites: which TEXCOORD indices GL_COORD_REPLACE rewrites, and the + * origin (PIPE_SPRITE_COORD_UPPER_LEFT or not). gl_PointCoord is always + * written. The vbuf widens points itself, so it writes the coordinates. */ +void sgx_vbuf_set_sprite(struct vbuf_render *r, unsigned coord_enable, + int upper_left); +/* Whether a point takes its size from the vertex program's gl_PointSize or + * from the rasterizer's constant (pipe_rasterizer_state::point_size_per_vertex; + * GL_VERTEX_PROGRAM_POINT_SIZE, always on under ES2). */ +void sgx_vbuf_set_psize_per_vertex(struct vbuf_render *r, int on); + +#endif diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_public.h mesa-26.2.2/src/gallium/drivers/sgx/sgx_public.h --- mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_public.h 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/sgx_public.h 2026-09-08 10:57:36.686120620 +0200 @@ -0,0 +1,13 @@ +/* SPDX-License-Identifier: MIT + * Copyright (C) 2026 RenĂ© Rebe */ +/* What the Mesa loader's screen constructor needs from the driver, and the + * only header of ours that drm_helper.h includes. It lives here rather than + * only inside a Mesa tree so that a sync into a fresh tree builds. */ +#ifndef SGX_PUBLIC_H +#define SGX_PUBLIC_H + +struct pipe_screen; + +struct pipe_screen *sgx_screen_create(int fd); + +#endif diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_resource.c mesa-26.2.2/src/gallium/drivers/sgx/sgx_resource.c --- mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_resource.c 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/sgx_resource.c 2026-09-08 10:57:36.680767706 +0200 @@ -0,0 +1,1165 @@ +/* Resources and transfers - see sgx_resource.h. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#include "sgx_resource.h" +#include "sgx_drm.h" +#include "xpsb_frame.h" + +#include +#include +#include +#include "sgx_context.h" /* sgx_log() */ + +/* A refused allocation reaches the caller as GL_OUT_OF_MEMORY and nothing + * else, so the size that was refused has to be reportable. */ +#define sgx_dbg(...) \ + do { if (getenv("SGX_DEBUG")) fprintf(sgx_log(), "sgx: " __VA_ARGS__); } \ + while (0) + +const char *sgx_resource_status_name(enum sgx_resource_status st) +{ + switch (st) { + case SGX_RESOURCE_OK: return "ok"; + case SGX_RESOURCE_BAD_FORMAT: return "format is not supported"; + case SGX_RESOURCE_BAD_SIZE: return "size out of range"; + case SGX_RESOURCE_BAD_BIND: return "format cannot serve that binding"; + case SGX_RESOURCE_NOMEM: return "out of memory"; + case SGX_RESOURCE_NOT_MAPPED: return "not mapped"; + } + return "?"; +} + +uint32_t sgx_resource_stride(enum sgx_format fmt, uint32_t width) +{ + unsigned bpp = xpsb_format_bpp((uint32_t)fmt); + + if (!bpp || !width) + return 0; + /* Not a choice: the descriptor has no pitch field, so this is what the + * hardware will read whatever the driver allocates. One definition, + * shared with the two surface checks that used to carry their own + * copy of it. */ + return xpsb_linear_stride((uint32_t)fmt, width); +} + +int sgx_resource_window(unsigned bind) +{ + /* Render targets and textures are both reached through the surface + * window; vertex and index data are not, and putting them there would + * bind them behind the wrong requestor. */ + if (bind & (SGX_BIND_SAMPLER_VIEW | SGX_BIND_RENDER_TARGET | + SGX_BIND_DEPTH_STENCIL | SGX_BIND_SCANOUT)) + return SGX_VM_SURFACE; + if (bind & (SGX_BIND_VERTEX_BUFFER | SGX_BIND_INDEX_BUFFER)) + return SGX_VM_RASTGEOM; + return -1; +} + +int sgx_resource_can_twiddle(uint32_t w, uint32_t h) +{ + return w && h && !(w & (w - 1u)) && !(h & (h - 1u)); +} + +static uint32_t next_pow2(uint32_t v) +{ + uint32_t n = 1; + + while (n < v) n <<= 1; + return n; +} + +/* The texel offset of a level from the start of its face; asked for level + * nlevels it is the texel count of the whole chain. Both rules are the + * vendor's, and the hardware derives every level's address from the base and + * the top level's size alone, so neither is a choice. + * + * Twiddled: GetMipMapOffset, gfx_linux_ddk opengles2/texdata.c:1740-1768. + * Levels are packed largest first with no padding, w*h texels each, both + * sides halved to a floor of one. The chain always runs to 1x1 + * (texmgmt.c:1741-1748) and level 0 sits at the base the descriptor names. + * + * Linear, the STRIDE texture type the descriptor's 0x60000000 selects: + * GetNPOTMipMapOffset, texdata.c:1857-2015, the vendor's own layout for a + * non-power-of-two chain (texmgmt.c:1773-1779). The top width is aligned to + * the 32-texel stride grain first (EURASIA_TAG_STRIDE_ALIGN1, sgxdefs.h:5039 + * for SGX_FEATURE_TEXTURE_STRIDE_GRANULARITY_32) and both sides are raised to + * a power of two; then every level's sides are aligned to 32 again before + * they are multiplied. The last step is what the blob's layout - the one this + * driver had - left out: it padded only the width, so a level whose height + * fell below 32 was placed too early and everything after it read garbage. + * Each level is still read at the pitch of its own width, ALIGN(w,32) + * (TranslateLevel, texdata.c:2210 and 2244-2251). */ +uint32_t sgx_resource_level_texels(uint32_t w0, uint32_t h0, int twiddled, + uint32_t level) +{ + uint32_t u = w0, v = h0, off = 0, i; + + if (!twiddled) { + u = next_pow2((w0 + 31u) & ~31u); + v = next_pow2(h0); + } + for (i = 0; i < level; i++) { + if (twiddled) + off += u * v; + else + off += ((u + 31u) & ~31u) * ((v + 31u) & ~31u); + u = u > 1u ? u >> 1 : 1u; + v = v > 1u ? v >> 1 : 1u; + } + return off; +} + +/* A volume's chain: each level w*h*d texels packed, every axis halving to a + * floor of one - the twiddled rule above with a third axis, which is not the + * vendor's (nothing in the vendor stack lays a volume out) but the only one + * consistent with it. */ +static uint32_t level_texels(const struct sgx_resource *r, uint32_t level) +{ + if (r->depth) + return xpsb_twiddle3_level_offset(r->width, r->height, + r->depth, level); + return sgx_resource_level_texels(r->width, r->height, + (int)r->twiddled, level); +} + +/* One plane's texel, and one CPU texel across the planes. */ +static unsigned plane_bpp(const struct sgx_resource *r) +{ + return xpsb_format_bpp((uint32_t)r->format); +} + +unsigned sgx_resource_texel_bytes(const struct sgx_resource *r) +{ + unsigned bb; + + if (!r) + return 0; + if (sgx_format_block(r->format, NULL, NULL, &bb)) + return bb; + return plane_bpp(r) * (r->nchunks ? r->nchunks : 1u); +} + +/* Bytes of one plane ahead of a level: texels at the plane's size, or + * blocks for a block format - a level is never smaller than one block, the + * vendor's GetCompressedMipMapOffset (opengles2/texdata.c:1779). */ +static uint32_t level_bytes_before(const struct sgx_resource *r, + unsigned bpp, uint32_t level) +{ + unsigned bw, bh, bb; + uint32_t w = r->width, h = r->height, off = 0, i; + + if (!sgx_format_block(r->format, &bw, &bh, &bb)) + return bpp * level_texels(r, level); + for (i = 0; i < level; i++) { + off += bb * ((w + bw - 1u) / bw) * ((h + bh - 1u) / bh); + w = w > 1u ? w >> 1 : 1u; + h = h > 1u ? h >> 1 : 1u; + } + return off; +} + +static int exact_log2(uint32_t v) +{ + int n = 0; + + if (!v || (v & (v - 1u))) + return -1; + while (v > 1u) { v >>= 1; n++; } + return n; +} + +static void layout_levels(struct sgx_resource *r, unsigned bpp) +{ + uint32_t w = r->width, h = r->height, d = r->depth ? r->depth : 1u; + uint32_t off, i; + + for (i = 0; i < r->nlevels; i++) { + r->level[i].width = w; + r->level[i].height = h; + r->level[i].depth = d; + /* The border map, when there is one, sits ahead of level 0 + * and the chain follows it, behind the low guard. */ + r->level[i].offset = r->border_guard + r->border_map + + level_bytes_before(r, bpp, i); + /* The pitch is never a choice: the descriptor has no pitch + * field, so the hardware reads a level at ALIGN(width,32) + * whatever is allocated. A twiddled level has none. */ + r->level[i].stride = r->twiddled ? 0u : + xpsb_linear_stride((uint32_t)r->format, w); + w = w > 1u ? w >> 1 : 1u; + h = h > 1u ? h >> 1 : 1u; + d = d > 1u ? d >> 1 : 1u; + } + off = r->border_guard + r->border_map + + level_bytes_before(r, bpp, r->nlevels); + /* A cube map is six of that chain in one object, and the hardware + * finds a face by multiplying the index by a stride it is never told - + * so the stride has to be the one the unit assumes. The DDK aligns it + * to 2048 bytes, but only for a mipmapped cube whose top level is + * above the per-format threshold; an unmipmapped one is packed. Get + * this wrong and every face but the first samples from a shifted + * address, which reads as five wrong faces rather than as a fault. */ + if (r->nfaces) { + uint32_t thresh = bpp == 1 ? SGX_CUBE_NO_ALIGN_8BPP : + SGX_CUBE_NO_ALIGN_WIDE; + + if (r->nlevels > 1 && r->width > thresh) + off = (off + (SGX_CUBE_FACE_ALIGN - 1u)) & + ~(SGX_CUBE_FACE_ALIGN - 1u); + r->face_stride = off; + off *= r->nfaces; + } + /* The high guard. A texture with a map has no faces, so this lands + * past the whole chain either way. */ + off += r->border_guard; + /* Diagnostic slack, in rows of the top level's stride. The hardware + * walks a linear chain at a stride this layout does not use, so a + * minified sample can compute an address past the object - see + * IOQUAKE3-PROFILE.md. Measuring how much slack stops the fault is + * what says how far past it reaches. */ + off += sgx_resource_slack_rows() * r->level[0].stride; + /* The raster pass writes past the last row, and sgx_resource_create() + * gives a linear render target thirty-two rows for it. A twiddled + * level has no row stride to count in, so the same allowance is made + * here in bytes off the top level's width. Nothing renders into a + * twiddled surface today - it is why generate_mipmap is advertised - + * but a texture carrying the bind can be handed to one, and an + * overrun outside the object corrupts whatever follows rather than + * drawing wrongly. It is a rounding on the smallest textures there + * are. */ + if (r->bind & SGX_BIND_RENDER_TARGET) + off += 32u * bpp * ((r->width + 31u) & ~31u); + /* Each plane is a whole texture of its own, the next starting on a + * texture address boundary (EURASIA_PDS_DOUTT2_TEXADDR_ALIGNSHIFT + * is 2; the vendor's chunks follow one another at the chain's size, + * texmgmt.c:1231). */ + if (r->nchunks > 1) + off = (off + 63u) & ~63u; + r->chunk_size = off; + r->size = off * (r->nchunks ? r->nchunks : 1u); +} + +/* Rows of top-level stride to over-allocate a sampled texture by. Zero unless + * SGX_TEX_SLACK_ROWS asks, so it costs nothing by default. */ +unsigned sgx_resource_slack_rows(void) +{ + static int rows = -1; + + if (rows < 0) { + const char *e = getenv("SGX_TEX_SLACK_ROWS"); + + rows = e && *e ? (int)strtoul(e, NULL, 0) : 0; + if (rows < 0) + rows = 0; + } + return (unsigned)rows; +} + +int sgx_border_ruler(void) +{ + const char *e = getenv("SGX_BORDER_RULER"); + + return e && *e && *e != '0'; +} + +unsigned sgx_border_guard_mult(void) +{ + const char *e = getenv("SGX_BORDER_GUARD"); + unsigned long n = e && *e ? strtoul(e, NULL, 0) : 0ul; + + if (n > 64ul) + n = 64ul; + return (unsigned)n; +} + +/* Not cached: the host suite sets SGX_BORDER around a case. */ +enum sgx_border_policy sgx_border_policy(void) +{ + const char *e = getenv("SGX_BORDER"); + + if (!e || !*e || *e == '0') + return SGX_BORDER_EDGE; + if (!strcmp(e, "fixed")) + return SGX_BORDER_FIXED; + if (!strcmp(e, "bdr1")) + return SGX_BORDER_BDR_BASE; + if (!strcmp(e, "bdr2")) + return SGX_BORDER_BDR_TEXELS; + if (*e == '2') + return SGX_BORDER_MAP_TEXELS; + return SGX_BORDER_MAP_BASE; +} + +int sgx_resource_border_maps(void) +{ + return sgx_border_policy() >= SGX_BORDER_MAP_BASE; +} + +enum xpsb_twiddle3 sgx_resource_twiddle3_order(void) +{ + static int order = -1; + + if (order < 0) { + const char *e = getenv("SGX_3D_LAYOUT"); + + order = e && !strcmp(e, "slices") ? XPSB_TWIDDLE3_SLICES + : XPSB_TWIDDLE3_MORTON; + } + return (enum xpsb_twiddle3)order; +} + +/* The one allocator behind the 2D, cube and volume entry points. faces is + * zero for an ordinary texture and six for a cube, depth zero except for a + * volume; they have to be threaded here rather than set by the caller because + * the descriptor is zeroed below, and a count set before the call would be + * the first thing lost. */ +static enum sgx_resource_status +create_tex_faces(struct sgx_resource *r, struct sgx_winsys *ws, + enum sgx_format fmt, uint32_t w, uint32_t h, + unsigned bind, uint32_t nlevels, int twiddled, + uint32_t faces, uint32_t depth, uint32_t nchunks) +{ + int window; + unsigned bpp, need = 0, block = sgx_format_block(fmt, NULL, NULL, NULL); + uint32_t max; + + if (!nlevels) + nlevels = 1; + if (!nchunks) + nchunks = 1; + if (!r || !ws || !w || !h) + return SGX_RESOURCE_BAD_SIZE; + /* A block format is twiddled at block granularity and nothing else: + * the descriptor has no pitch for it and the vendor never lays one + * out linearly (texdata.c DeTwiddleAddressETC1). */ + if (block) + twiddled = 1; + if (nlevels == 1 && !twiddled && !faces && !depth && nchunks == 1) + return sgx_resource_create(r, ws, fmt, w, h, bind); + if (twiddled && !sgx_resource_can_twiddle(w, h)) + return SGX_RESOURCE_BAD_SIZE; + /* A volume has only the log2 encoding, in every axis. */ + if (depth && (!twiddled || faces || (depth & (depth - 1u)) || + depth > (uint32_t)sgx_screen_cap(SGX_CAP_MAX_TEXTURE_SIZE))) + return SGX_RESOURCE_BAD_SIZE; + + window = sgx_resource_window(bind); + if (window < 0) + return SGX_RESOURCE_BAD_BIND; + if (bind & SGX_BIND_SAMPLER_VIEW) + need |= SGX_FMT_BIND_SAMPLER; + if (bind & SGX_BIND_RENDER_TARGET) + need |= SGX_FMT_BIND_RENDER; + if (!need || !sgx_format_supported(fmt, need)) + return SGX_RESOURCE_BAD_BIND; + bpp = xpsb_format_bpp((uint32_t)fmt); + if (!bpp && !block) + return SGX_RESOURCE_BAD_FORMAT; + if (w > (uint32_t)sgx_screen_cap(SGX_CAP_MAX_TEXTURE_SIZE) || + h > (uint32_t)sgx_screen_cap(SGX_CAP_MAX_TEXTURE_SIZE)) + return SGX_RESOURCE_BAD_SIZE; + /* A chain longer than the size allows would place levels the hardware + * never reads, and the top level index has only four bits. */ + for (max = 1; (w >> max) || (h >> max) || (depth >> max); max++) + ; + if (nlevels > max || nlevels > SGX_MAX_LEVELS || nlevels > 15u) + return SGX_RESOURCE_BAD_SIZE; + + memset(r, 0, sizeof(*r)); + r->format = fmt; + r->width = w; + r->height = h; + r->bind = bind; + r->stride = sgx_resource_stride(fmt, w); + r->nlevels = nlevels; + r->twiddled = twiddled ? 1u : 0u; + r->nfaces = faces; + r->depth = depth; + r->nchunks = nchunks; + r->twiddle3 = (uint32_t)sgx_resource_twiddle3_order(); + /* The border map is the DDK's square table, in the texture's own + * format, ahead of the texels; a shape it has no entry for gets none + * and the border modes are refused on it. */ + if (sgx_resource_border_maps() && twiddled && !faces && !depth && + !block && nchunks == 1 && w == h && (bind & SGX_BIND_SAMPLER_VIEW)) { + /* The face offset, not the table's whole extent. Handed the + * map's base, the unit fetches at base + OFFSET_x - + * measured 2026-09-03 at 2x2, 4x4 and 8x8, the last to the + * byte - so that is where level 0 has to be. Laying the whole + * extent in front put it 8n texels too far on, which is what + * every border run until now was measuring. */ + r->border_map = bpp * xpsb_border_map_face_offset(w); + /* The unit displaces its whole fetch under a border mode, by + * an amount and a sign that are what is being measured. The + * guard is what that lands in instead of in another object - + * several map-sizes of it, because at exactly one the two + * candidate displacements land on a boundary and the run that + * hit that could not tell them apart. */ + r->border_guard = sgx_border_guard_mult() * r->border_map; + } + /* Laid out before anything is allocated, so the chain is allocated + * once at its real extent rather than allocated flat and grown. */ + layout_levels(r, bpp); + if (!r->size || r->size > SGX_RESOURCE_MAX_BYTES) + return SGX_RESOURCE_BAD_SIZE; + + if (sgx_bo_new(ws, &r->bo, r->size, + (bind & SGX_BIND_SCANOUT) ? SGX_BO_SCANOUT : 0)) { + sgx_dbg("tex: %u levels of %ux%u wants %u bytes, refused\n", + nlevels, w, h, r->size); + memset(r, 0, sizeof(*r)); + return SGX_RESOURCE_NOMEM; + } + if (sgx_bo_bind(ws, &r->bo, (uint32_t)window)) { + sgx_dbg("tex: %u bytes allocated but would not bind\n", + r->size); + sgx_bo_free(ws, &r->bo); + memset(r, 0, sizeof(*r)); + return SGX_RESOURCE_NOMEM; + } + r->bound = 1; + /* The layout a border run is measured against. An MMU fault names an + * address and nothing else, so without this line it cannot be turned + * into a displacement - which is what the first fault cost. */ + if (r->border_map) + sgx_dbg("border: %ux%u at 0x%08x, guard %u, map %u, " + "level 0 0x%08x, %u bytes\n", w, h, + (uint32_t)r->bo.gpu_va, r->border_guard, r->border_map, + (uint32_t)r->bo.gpu_va + r->level[0].offset, r->size); + return SGX_RESOURCE_OK; +} + +enum sgx_resource_status +sgx_resource_create_tex(struct sgx_resource *r, struct sgx_winsys *ws, + enum sgx_format fmt, uint32_t w, uint32_t h, + unsigned bind, uint32_t nlevels, int twiddled) +{ + return create_tex_faces(r, ws, fmt, w, h, bind, nlevels, twiddled, 0u, + 0u, 1u); +} + +enum sgx_resource_status +sgx_resource_create_planes(struct sgx_resource *r, struct sgx_winsys *ws, + enum sgx_format fmt, uint32_t w, uint32_t h, + unsigned bind, uint32_t nlevels, int twiddled, + uint32_t nchunks) +{ + return create_tex_faces(r, ws, fmt, w, h, bind, nlevels, twiddled, 0u, + 0u, nchunks); +} + +enum sgx_resource_status +sgx_resource_create_3d(struct sgx_resource *r, struct sgx_winsys *ws, + enum sgx_format fmt, uint32_t w, uint32_t h, uint32_t d, + unsigned bind, uint32_t nlevels) +{ + if (!r || !d) + return SGX_RESOURCE_BAD_SIZE; + return create_tex_faces(r, ws, fmt, w, h, bind, nlevels, 1, 0u, d, 1u); +} + +/* One side of the guard, as a ruler: each texel names the 64-byte block of + * the object it lies in, so a probe that reads one says how far the unit's + * fetch was displaced and in which direction. */ +static void fill_guard(struct sgx_resource *r, void *p, uint32_t base, + uint32_t bytes, unsigned bpp, unsigned side) +{ + uint32_t i; + + for (i = 0; i + bpp <= bytes; i += bpp) { + uint32_t blk = (base + i) / SGX_BORDER_GUARD_BLOCK; + float rgba[4]; + + rgba[0] = (float)(blk & 0xffu) / 255.0f; + rgba[1] = (float)((blk >> 8) & 0xffu) / 255.0f; + rgba[2] = (float)side / 255.0f; + rgba[3] = 1.0f; + sgx_format_pack(r->format, rgba, (unsigned char *)p + base + i); + } +} + +enum sgx_resource_status +sgx_resource_fill_border(struct sgx_resource *r, struct sgx_winsys *ws, + const float rgba[4]) +{ + enum sgx_resource_status st; + unsigned char texel[4]; + unsigned bpp, i; + uint32_t hi; + void *p = NULL; + + if (!r || !rgba) + return SGX_RESOURCE_NOT_MAPPED; + if (!r->border_map) + return SGX_RESOURCE_BAD_BIND; + if (r->border_filled && !memcmp(r->border_rgba, rgba, sizeof r->border_rgba)) + return SGX_RESOURCE_OK; + bpp = sgx_format_pack(r->format, rgba, texel); + if (!bpp || bpp != xpsb_format_bpp((uint32_t)r->format)) + return SGX_RESOURCE_BAD_FORMAT; + st = sgx_resource_map(r, ws, &p); + if (st != SGX_RESOURCE_OK) + return st; + fill_guard(r, p, 0u, r->border_guard, bpp, SGX_BORDER_GUARD_LOW_B); + /* The border colour, or the map's own ruler when a run is asking + * where the unit takes its border from rather than what colour it + * is. */ + if (sgx_border_ruler()) + fill_guard(r, p, r->border_guard, r->border_map, bpp, + SGX_BORDER_MAP_B); + else + for (i = 0; i < r->border_map; i += bpp) + memcpy((char *)p + r->border_guard + i, texel, bpp); + hi = r->border_guard + r->border_map + + level_bytes_before(r, bpp, r->nlevels); + fill_guard(r, p, hi, + hi + r->border_guard <= r->size ? r->border_guard : + r->size - hi, + bpp, SGX_BORDER_GUARD_HIGH_B); + sgx_resource_unmap(r); + memcpy(r->border_rgba, rgba, sizeof r->border_rgba); + r->border_filled = 1; + return SGX_RESOURCE_OK; +} + +/* Six faces of one chain in one object. The unit takes a single base address + * and finds a face by multiplying its index by a stride nothing tells it, so + * the layout here has to be the one it assumes: face-major, each face the + * whole mip chain, the faces of a mipmapped cube aligned to 2048 bytes. + * + * Twiddled, always: the cube texture type is one of the log2-size encodings, + * which is the twiddled layout. There is no linear cube. */ +enum sgx_resource_status +sgx_resource_create_cube(struct sgx_resource *r, struct sgx_winsys *ws, + enum sgx_format fmt, uint32_t size, + unsigned bind, uint32_t nlevels) +{ + if (!r) + return SGX_RESOURCE_BAD_SIZE; + /* A face is described by log2 of one side, so a cube that is not + * square or not a power of two has no encoding at all. */ + if (!size || (size & (size - 1u))) + return SGX_RESOURCE_BAD_SIZE; + return create_tex_faces(r, ws, fmt, size, size, bind, nlevels, 1, + SGX_CUBE_FACES, 0u, 1u); +} + +uint32_t sgx_resource_face_offset(const struct sgx_resource *r, + uint32_t face, uint32_t level) +{ + uint32_t at; + + if (!r || level >= (r->nlevels ? r->nlevels : 1u)) + return 0; + at = r->nlevels ? r->level[level].offset : 0u; + if (r->nfaces && face < r->nfaces) + at += face * r->face_stride; + return at; +} + +/* Whether the CPU sees a level through a linear staging copy: a twiddled + * level cannot be written a row at a time, a plane-stored texel is not + * contiguous, and a block format is both. */ +static int needs_stage(const struct sgx_resource *r) +{ + return r->twiddled || r->nchunks > 1 || + sgx_format_block(r->format, NULL, NULL, NULL); +} + +/* The planes and blocks take the copy below; a volume, and a plain + * twiddled 2D level, keep the twiddle helpers. */ +static int stage_by_copy(const struct sgx_resource *r) +{ + return !r->depth && (r->nchunks > 1 || + sgx_format_block(r->format, NULL, NULL, NULL)); +} + +/* The staging copy's row, in bytes: texels interleaved across the planes, + * or a row of blocks. */ +static uint32_t stage_stride(const struct sgx_resource *r, uint32_t w) +{ + unsigned bw, bb; + + if (sgx_format_block(r->format, &bw, NULL, &bb)) + return bb * ((w + bw - 1u) / bw); + return sgx_resource_texel_bytes(r) * w; +} + +static uint32_t stage_bytes(const struct sgx_resource *r, uint32_t w, + uint32_t h) +{ + unsigned bh; + + if (sgx_format_block(r->format, NULL, &bh, NULL)) + h = (h + bh - 1u) / bh; + return stage_stride(r, w) * h; +} + +/* Move one face's level between the object and the staging copy: plane by + * plane, texel by texel, through the twiddle when the level is twiddled. A + * block format moves blocks over the block grid, and each block's two dwords + * go out byte-reversed - the vendor's CopyTextureETC1 (opengles2/tex.c:558, + * "HW byte ordering is other-endian"). */ +static void stage_copy(struct sgx_resource *r, void *obj, uint32_t face, + uint32_t level, int to_object) +{ + const struct sgx_resource_level *lv = &r->level[level]; + unsigned bw, bh, bb, bpp = plane_bpp(r); + unsigned nch = r->nchunks ? r->nchunks : 1u, unit; + uint32_t w = lv->width, h = lv->height, x, y, k, sstride; + int lw, lh, block = sgx_format_block(r->format, &bw, &bh, &bb) != 0; + unsigned char *plane = (unsigned char *)obj + + sgx_resource_face_offset(r, face, level); + unsigned char *st = r->stage; + + sstride = stage_stride(r, w); + if (block) { + w = (w + bw - 1u) / bw; + h = (h + bh - 1u) / bh; + unit = bb; + nch = 1; + } else { + unit = bpp; + } + lw = exact_log2(w); + lh = exact_log2(h); + for (k = 0; k < nch; k++, plane += r->chunk_size) { + for (y = 0; y < h; y++) { + for (x = 0; x < w; x++) { + unsigned char *a, *b = st + (size_t)y * sstride + + (size_t)x * (block ? unit : bpp * nch) + + (block ? 0u : k * bpp); + + if (r->twiddled && lw >= 0 && lh >= 0) + a = plane + (size_t)unit * + xpsb_twiddle_index(x, y, (uint32_t)lw, + (uint32_t)lh); + else + a = plane + (size_t)y * lv->stride + + (size_t)x * unit; + if (block) { + unsigned i; + + for (i = 0; i < unit; i++) { + unsigned j = (i & ~3u) | + (3u - (i & 3u)); + + if (to_object) + a[j] = b[i]; + else + b[i] = a[j]; + } + } else if (to_object) { + memcpy(a, b, unit); + } else { + memcpy(b, a, unit); + } + } + } + } +} + +static enum sgx_resource_status +sgx_resource_map_face_level_mode(struct sgx_resource *r, struct sgx_winsys *ws, + uint32_t face, uint32_t level, void **out, + uint32_t *stride, int ro) +{ + enum sgx_resource_status st; + unsigned bpp; + uint32_t at; + void *p = NULL; + + if (!r || !out || level >= (r->nlevels ? r->nlevels : 1u)) + return SGX_RESOURCE_NOT_MAPPED; + if (face >= (r->nfaces ? r->nfaces : 1u)) + return SGX_RESOURCE_NOT_MAPPED; + /* A reader whose staging copy is already good needs nothing from the + * object itself, and must not ask for it: sgx_resource_map() waits for + * the part to go idle, so mapping a vertex texture once per draw - the + * draw module does exactly that - serialised the CPU against the GPU + * for the whole scene. That wait is what ioread32 and delay_tsc were + * doing at the top of terrain's profile. */ + if (ro && r->nlevels && needs_stage(r) && r->stage && r->stage_valid && + r->stage_level == level && r->stage_face == face) { + unsigned b = xpsb_format_bpp((uint32_t)r->format); + + if (stride) *stride = stage_by_copy(r) ? + stage_stride(r, r->level[level].width) : + b * r->level[level].width; + *out = r->stage; + return SGX_RESOURCE_OK; + } + st = sgx_resource_map(r, ws, &p); + if (st != SGX_RESOURCE_OK) + return st; + if (!r->nlevels) { + if (stride) *stride = r->stride; + *out = p; + return SGX_RESOURCE_OK; + } + at = sgx_resource_face_offset(r, face, level); + if (!needs_stage(r)) { + if (stride) *stride = r->level[level].stride; + *out = (char *)p + at; + return SGX_RESOURCE_OK; + } + + bpp = xpsb_format_bpp((uint32_t)r->format); + { + uint32_t need = bpp * r->width * r->height * r->level[0].depth; + uint32_t need2 = stage_bytes(r, r->width, r->height); + + if (need2 > need) + need = need2; + if (r->stage_size < need) { + free(r->stage); + r->stage_size = need; + r->stage = NULL; + } + } + if (!r->stage) { + r->stage = malloc(r->stage_size); + if (!r->stage) { + r->stage_size = 0; + sgx_resource_unmap(r); + return SGX_RESOURCE_NOMEM; + } + } + /* Already untwiddled, and nothing has written the resource since, so + * it can be handed over as it stands - which is the whole of what + * makes a vertex texture fetch affordable, since the draw module maps + * its textures once per draw. */ + if (r->stage_valid && r->stage_level == level && + r->stage_face == face) { + if (!ro) + r->stage_dirty = 1; + if (stride) *stride = stage_by_copy(r) ? + stage_stride(r, r->level[level].width) : + bpp * r->level[level].width; + *out = r->stage; + return SGX_RESOURCE_OK; + } + /* The staging copy starts out as what is already there. Without this a + * caller that writes part of a level would twiddle undefined bytes + * over the rest of it, and one that only reads would see nothing. */ + if (stage_by_copy(r)) + stage_copy(r, p, face, level, 0); + else if (r->depth) + xpsb_untwiddle3_level(r->stage, (char *)p + at, + r->level[level].width, + r->level[level].height, + r->level[level].depth, + bpp * r->level[level].width, + bpp * r->level[level].width * + r->level[level].height, bpp, + (enum xpsb_twiddle3)r->twiddle3); + else + xpsb_untwiddle_level(r->stage, (char *)p + at, + r->level[level].width, + r->level[level].height, + bpp * r->level[level].width, bpp); + r->stage_level = level; + r->stage_face = face; + r->stage_valid = 1; + r->stage_dirty = !ro; + if (stride) *stride = stage_by_copy(r) ? + stage_stride(r, r->level[level].width) : + bpp * r->level[level].width; + *out = r->stage; + return SGX_RESOURCE_OK; +} + +enum sgx_resource_status +sgx_resource_map_face_level(struct sgx_resource *r, struct sgx_winsys *ws, + uint32_t face, uint32_t level, void **out, + uint32_t *stride) +{ + return sgx_resource_map_face_level_mode(r, ws, face, level, out, + stride, 0); +} + +enum sgx_resource_status +sgx_resource_map_level(struct sgx_resource *r, struct sgx_winsys *ws, + uint32_t level, void **out, uint32_t *stride) +{ + return sgx_resource_map_face_level(r, ws, 0u, level, out, stride); +} + +enum sgx_resource_status +sgx_resource_map_level_ro(struct sgx_resource *r, struct sgx_winsys *ws, + uint32_t level, void **out, uint32_t *stride) +{ + return sgx_resource_map_face_level_mode(r, ws, 0u, level, out, stride, + 1); +} + +enum sgx_resource_status sgx_resource_unmap_level(struct sgx_resource *r) +{ + if (!r) + return SGX_RESOURCE_NOT_MAPPED; + if (needs_stage(r) && r->stage_dirty && r->map) { + unsigned bpp = xpsb_format_bpp((uint32_t)r->format); + uint32_t l = r->stage_level; + + char *at = (char *)r->map + + sgx_resource_face_offset(r, r->stage_face, l); + + if (stage_by_copy(r)) + stage_copy(r, r->map, r->stage_face, l, 1); + else if (r->depth) + xpsb_twiddle3_level(at, r->stage, r->level[l].width, + r->level[l].height, + r->level[l].depth, + bpp * r->level[l].width, + bpp * r->level[l].width * + r->level[l].height, bpp, + (enum xpsb_twiddle3)r->twiddle3); + else + xpsb_twiddle_level(at, r->stage, r->level[l].width, + r->level[l].height, + bpp * r->level[l].width, bpp); + r->stage_dirty = 0; + r->stage_valid = 0; + } + return sgx_resource_unmap(r); +} + +/* Bytes a depth surface needs, at a sample count. + * + * The colour target does not grow with the sample count and this does. The + * part resolves colour on store: the vendor's EGL allocates the drawable's + * colour buffer at the drawable's size whether or not the surface is + * multisampled and records the render target as SGX_DOWNSCALING instead + * (srv_sgx.c:485-537, sgxrender_targets.c:1067-1091), while the same code + * multiplies the depth allocation by four and the ZLS extent by two for a 2x2 + * surface (srv_sgx.c:704-713, 410-414). + * + * Zero for a sample count the part cannot reach, so a caller cannot be handed + * a buffer sized for a mode that will not run. */ +uint32_t sgx_resource_depth_bytes(uint32_t w, uint32_t h, unsigned samples) +{ + uint32_t stride = 4u * ((w + 31u) & ~31u); + uint32_t axis = xpsb_msaa_axis(samples ? samples : 1u); + uint32_t rows; + + if (!axis || !w || !h) + return 0; + /* The raster pass writes past the last row, so the same 32 rows of + * slack the target gets - at the sample resolution the ISP stores. */ + rows = (h + 32u) * axis; + if (stride * axis > SGX_RESOURCE_MAX_BYTES / rows) + return 0; + return stride * axis * rows; +} + +enum sgx_resource_status +sgx_resource_create_ms(struct sgx_resource *r, struct sgx_winsys *ws, + enum sgx_format fmt, uint32_t w, uint32_t h, + unsigned bind, unsigned samples) +{ + int window; + int ret; + + if (!r || !ws || !w || !h) + return SGX_RESOURCE_BAD_SIZE; + if (!xpsb_msaa_axis(samples)) + return SGX_RESOURCE_BAD_SIZE; + window = sgx_resource_window(bind); + if (window < 0) + return SGX_RESOURCE_BAD_BIND; + + memset(r, 0, sizeof(*r)); + r->format = fmt; + r->width = w; + r->height = h; + r->samples = samples; + r->bind = bind; + + if (bind & SGX_BIND_DEPTH_STENCIL) { + if (w > (uint32_t)sgx_screen_cap(SGX_CAP_MAX_TEXTURE_SIZE) || + h > (uint32_t)sgx_screen_cap(SGX_CAP_MAX_TEXTURE_SIZE)) + return SGX_RESOURCE_BAD_SIZE; + r->stride = 4u * ((w + 31u) & ~31u); + r->size = sgx_resource_depth_bytes(w, h, samples); + if (!r->size) + return SGX_RESOURCE_BAD_SIZE; + } else if (fmt == SGX_FMT_NONE) { + /* A plain buffer. Nothing reads it through a surface + * descriptor, so the row padding does not apply. */ + if (bind & (SGX_BIND_SAMPLER_VIEW | SGX_BIND_RENDER_TARGET)) + return SGX_RESOURCE_BAD_BIND; + /* Nothing bounds a buffer's dimensions the way a texture's are + * bounded, so w * h can wrap. It wraps to a small number, the + * allocation succeeds, and the caller believes it has the + * buffer it asked for - which is worse than failing. */ + if (h && w > SGX_RESOURCE_MAX_BYTES / h) + return SGX_RESOURCE_BAD_SIZE; + r->stride = w; + r->size = w * h; + } else { + unsigned need = 0; + + if (bind & SGX_BIND_SAMPLER_VIEW) + need |= SGX_FMT_BIND_SAMPLER; + if (bind & SGX_BIND_RENDER_TARGET) + need |= SGX_FMT_BIND_RENDER; + if (!need) + return SGX_RESOURCE_BAD_BIND; + /* Checked against every binding at once: a format that can be + * sampled but not rendered must fail for a resource that is + * both, not succeed on the strength of the first. */ + if (!sgx_format_supported(fmt, need)) + return SGX_RESOURCE_BAD_BIND; + + if (w > (uint32_t)sgx_screen_cap(SGX_CAP_MAX_TEXTURE_SIZE) || + h > (uint32_t)sgx_screen_cap(SGX_CAP_MAX_TEXTURE_SIZE)) + return SGX_RESOURCE_BAD_SIZE; + + r->stride = sgx_resource_stride(fmt, w); + if (!r->stride) + return SGX_RESOURCE_BAD_FORMAT; + /* The size limits above make this safe today - 4 bytes times + * 2048 padded times 2048 is 16 MiB - but the check is here + * rather than in a comment, because the limits are a table + * someone will edit. */ + if (h && r->stride > SGX_RESOURCE_MAX_BYTES / h) + return SGX_RESOURCE_BAD_SIZE; + r->size = r->stride * h; + /* The raster pass writes past the last row - sgxtri gives the + * target 32 rows of slack and so does the context's own + * allocation, so one the caller supplies needs the same or the + * overrun lands outside the object. */ + if (bind & SGX_BIND_RENDER_TARGET) + r->size = r->stride * (h + 32); + else if (bind & SGX_BIND_SAMPLER_VIEW) + r->size += sgx_resource_slack_rows() * r->stride; + } + + ret = sgx_bo_new(ws, &r->bo, r->size, + (bind & SGX_BIND_SCANOUT) ? SGX_BO_SCANOUT : 0); + if (ret) + return SGX_RESOURCE_NOMEM; + /* One level, described the way a chain describes its first. Every + * reader that asks a resource about level zero rather than about the + * resource - the draw module's vertex textures do, through + * sgx_pipe_map_vtx_textures() - was handed a width and height of zero + * here, because only the multi-level path filled this in. A zero-sized + * texture clamps every coordinate to its first texel, which is why + * glmark2's terrain sampled one heightmap texel for every vertex and + * drew a flat plane. */ + if (!r->nlevels) { + r->nlevels = 1; + r->level[0].width = r->width; + r->level[0].height = r->height; + r->level[0].depth = r->depth > 1 ? r->depth : 1; + r->level[0].stride = r->stride; + r->level[0].offset = 0; + } + + /* Vertex and index data never reach the part through an address of + * their own: the context reads them with the CPU and copies what a + * draw needs into the frame's own vertex object. Binding them anyway + * spent the raster-geometry window, which the 24-bit offsets the + * relocations carry hold to sixteen megabytes - Quake 3 ran out of + * buffers there after fifteen and three quarters. An object with no + * address still maps, which is all these need. */ + if (window == SGX_VM_RASTGEOM && + !(bind & (SGX_BIND_SAMPLER_VIEW | SGX_BIND_RENDER_TARGET | + SGX_BIND_DEPTH_STENCIL | SGX_BIND_SCANOUT))) { + r->bound = 0; + return SGX_RESOURCE_OK; + } + ret = sgx_bo_bind(ws, &r->bo, (uint32_t)window); + if (ret) { + /* The object exists even though the bind failed, so it has to + * be closed here. Returning without this leaks it, and the + * caller has no handle to close it with. */ + sgx_bo_free(ws, &r->bo); + return SGX_RESOURCE_NOMEM; + } + r->bound = 1; + return SGX_RESOURCE_OK; +} + +enum sgx_resource_status +sgx_resource_create(struct sgx_resource *r, struct sgx_winsys *ws, + enum sgx_format fmt, uint32_t w, uint32_t h, unsigned bind) +{ + return sgx_resource_create_ms(r, ws, fmt, w, h, bind, 1u); +} + +/* Lay a resource out linearly, keeping what it holds. + * + * The raster pass writes a linear surface whatever the resource says, so a + * twiddled one that is rendered into is written one way and sampled another - + * the texels come back moved about in blocks. Nothing can be decided at + * creation: Mesa gives every colour-renderable texture the render-target bind + * whether or not the caller ever attaches it, so a mipmapped texture and an + * FBO attachment arrive here identical. This is called the first time one is + * really attached. + * + * The chain comes with it, read through the staging path that de-twiddles and + * written back into the linear layout. A linear chain is not described to the + * unit, so the texture samples its base level from here on - correct texels, + * no minification quality, which is the trade the linear layout already makes + * everywhere else. + */ +enum sgx_resource_status +sgx_resource_make_linear(struct sgx_resource *r, struct sgx_winsys *ws) +{ + struct sgx_resource n; + enum sgx_resource_status st; + unsigned char **keep; + uint32_t *keep_stride; + uint32_t l, nl; + int ok = 1; + + if (!r || !ws) + return SGX_RESOURCE_BAD_SIZE; + if (!r->twiddled || r->nfaces > 1 || r->depth > 1 || r->nchunks > 1) + return SGX_RESOURCE_OK; /* nothing to do, or not a form this moves */ + nl = r->nlevels ? r->nlevels : 1u; + keep = calloc(nl, sizeof *keep); + keep_stride = calloc(nl, sizeof *keep_stride); + if (!keep || !keep_stride) { + free(keep); + free(keep_stride); + return SGX_RESOURCE_NOMEM; + } + /* Read the whole chain out before the object goes: one level at a + * time, because a twiddled resource has a single staging buffer. */ + for (l = 0; l < nl; l++) { + unsigned char *src = NULL; + uint32_t stride = 0; + size_t bytes; + + if (sgx_resource_map_face_level(r, ws, 0u, l, (void **)&src, + &stride) != SGX_RESOURCE_OK) { + ok = 0; + break; + } + if (!stride) + stride = sgx_format_bytes(r->format) * r->level[l].width; + bytes = (size_t)stride * r->level[l].height; + keep[l] = malloc(bytes); + if (keep[l]) + memcpy(keep[l], src, bytes); + else + ok = 0; + keep_stride[l] = stride; + sgx_resource_unmap_level(r); + if (!ok) + break; + } + + memset(&n, 0, sizeof n); + st = sgx_resource_create_planes(&n, ws, r->format, r->width, r->height, + r->bind, nl, 0, 1u); + if (st != SGX_RESOURCE_OK) { + for (l = 0; l < nl; l++) + free(keep[l]); + free(keep); + free(keep_stride); + return st; + } + if (ok) { + for (l = 0; l < nl; l++) { + unsigned char *dst = NULL; + uint32_t dstride = 0; + uint32_t y, row; + + if (!keep[l] || + sgx_resource_map_face_level(&n, ws, 0u, l, + (void **)&dst, + &dstride) != + SGX_RESOURCE_OK) + continue; + if (!dstride) + dstride = sgx_format_bytes(n.format) * + n.level[l].width; + row = dstride < keep_stride[l] ? dstride : + keep_stride[l]; + for (y = 0; y < n.level[l].height; y++) + memcpy(dst + (size_t)y * dstride, + keep[l] + (size_t)y * keep_stride[l], + row); + sgx_resource_unmap_level(&n); + } + } + for (l = 0; l < nl; l++) + free(keep[l]); + free(keep); + free(keep_stride); + + sgx_resource_destroy(r, ws); + *r = n; + return SGX_RESOURCE_OK; +} + +void sgx_resource_destroy(struct sgx_resource *r, struct sgx_winsys *ws) +{ + if (!r) + return; + /* Unbind and close, rather than just forgetting the struct. Forgetting + * it leaks the object in the kernel, where the binding holds a + * reference that nothing else will ever drop. */ + if (r->bo.handle) { + if (r->foreign) + sgx_bo_unbind(ws, &r->bo); + else + sgx_bo_free(ws, &r->bo); + } + free(r->stage); + free(r->shadow); + memset(r, 0, sizeof(*r)); +} + +enum sgx_resource_status +sgx_resource_map_nowait(struct sgx_resource *r, struct sgx_winsys *ws, + void **out, int nowait) +{ + /* On the object existing, not on it having an address: a render + * target has no address until it is the bound framebuffer's, and is + * mappable throughout. */ + if (!r || !ws || !out || !r->bo.handle) + return SGX_RESOURCE_NOT_MAPPED; + /* Every CPU access to a resource comes through here - a transfer map, + * a subdata upload, a copy, an export - so this is where a deferred + * render is waited for. Reading before it has ended returns the frame + * before; writing into a texture it is still sampling corrupts the + * frame on the core. A no-op unless a frame is outstanding. + * + * Unless the caller says it does not need one. PIPE_MAP_UNSYNCHRONIZED + * is a promise that what is about to be written is not what the part + * is reading - Mesa's stream uploader appends into a region it has + * just taken, and every upload of a client array is one of those. The + * wait is not merely wasted there: it drains the part on every upload, + * which is a frame's worth of serialisation for nothing. */ + if (!nowait) + sgx_wait_idle(ws); + if (!r->map) { + if (sgx_bo_map(ws, &r->bo)) + return SGX_RESOURCE_NOMEM; + r->map = r->bo.map; + } + r->map_count++; + *out = r->map; + return SGX_RESOURCE_OK; +} + +enum sgx_resource_status +sgx_resource_map(struct sgx_resource *r, struct sgx_winsys *ws, void **out) +{ + return sgx_resource_map_nowait(r, ws, out, 0); +} + +enum sgx_resource_status sgx_resource_unmap(struct sgx_resource *r) +{ + if (!r || !r->map_count) + return SGX_RESOURCE_NOT_MAPPED; + r->map_count--; + return SGX_RESOURCE_OK; +} diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_resource.h mesa-26.2.2/src/gallium/drivers/sgx/sgx_resource.h --- mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_resource.h 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/sgx_resource.h 2026-09-08 10:57:36.681102182 +0200 @@ -0,0 +1,350 @@ +/* Resources and transfers: allocating a texture or a render target, and + * getting CPU access to it. + * + * The interesting part is not the allocation, it is the layout. The surface + * descriptor carries a width and a format, not a pitch, so the hardware works + * out where each row starts - which means a driver does not get to choose the + * stride. A resource whose rows are packed tightly is read as though they were + * padded, so the padding is a property of the resource and is computed here, + * once, rather than assumed at each use. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#ifndef _SGX_RESOURCE_H_ +#define _SGX_RESOURCE_H_ + +#include + +#include "sgx_screen.h" +#include "sgx_winsys.h" + +/* pipe_resource::bind, the subset that changes the layout or the placement. */ +#define SGX_BIND_SAMPLER_VIEW (1u << 0) +#define SGX_BIND_RENDER_TARGET (1u << 1) +#define SGX_BIND_VERTEX_BUFFER (1u << 2) +#define SGX_BIND_INDEX_BUFFER (1u << 3) +/* The depth buffer. It has no entry in the format table - the frame owns it + * and never describes it with a surface descriptor - so it is laid out here by + * what the hardware writes rather than by a pixel format: four bytes a pixel, + * the same row padding as a render target, and the same rows of slack. */ +#define SGX_BIND_DEPTH_STENCIL (1u << 4) +/* The display engine will scan this out. It changes which kind of object the + * kernel allocates, not just where it goes: only gma500's own GEM objects + * carry the GTT resource the display pins, so this cannot be decided after + * the fact. */ +#define SGX_BIND_SCANOUT (1u << 5) + +/* The largest object worth allowing. The surface window is 128 MiB and the + * parameter window 254 MiB, so nothing legitimate approaches this; it exists so + * that a size computation cannot wrap. */ +#define SGX_RESOURCE_MAX_BYTES 0x08000000u + +/* 2048 is the largest texture, so twelve levels reach 1x1; one more so a + * loop that writes level nlevels has somewhere to stop. */ +#define SGX_MAX_LEVELS 13 + +struct sgx_resource_level { + uint32_t offset; /* bytes from the start of the object */ + uint32_t width, height; + uint32_t depth; /* slices; 1 for a 2D level */ + uint32_t stride; /* 0 for a twiddled level, which has no + * pitch at all */ +}; + +/* A cube map's six faces, in the order GL numbers them: +X -X +Y -Y +Z -Z. + * The hardware takes one base address for the whole texture and finds a face + * by multiplying this index by the face stride, so the order is the layout. */ +#define SGX_CUBE_FACES 6u +/* Faces of a mipmapped cube start on a 2048-byte boundary - the DDK's + * EURASIA_TAG_CUBEMAP_FACE_ALIGN - unless the top level is small enough not to + * need it. The threshold is per format: sixteen texels at one byte a texel, + * eight at two or four. An unmipmapped cube is packed with no alignment. */ +#define SGX_CUBE_FACE_ALIGN 2048u +#define SGX_CUBE_NO_ALIGN_8BPP 16u +#define SGX_CUBE_NO_ALIGN_WIDE 8u + +struct sgx_resource { + struct sgx_bo bo; + enum sgx_format format; /* SGX_FMT_NONE for a plain buffer */ + uint32_t width, height; + /* Samples a pixel: 1, or 4 for the part's 2x2 mode. Only a depth + * surface grows with it - the pixel back end downscales colour on + * store, so a multisampled colour target is the same memory a single + * sampled one is. */ + uint32_t samples; + uint32_t stride; /* bytes per row, as the hardware reads */ + uint32_t size; + uint32_t nlevels; /* 1 unless there is a mip chain */ + uint32_t twiddled; + /* Six for a cube map, zero for everything else - not one, so that + * "is this a cube" is the field itself rather than a comparison. */ + uint32_t nfaces; + /* Bytes from one face's level 0 to the next. Zero unless nfaces. */ + uint32_t face_stride; + /* Slices of a volume, zero for everything else, in the 3D twiddle + * order twiddle3 (enum xpsb_twiddle3) - which order the unit walks is + * not in the DDK, so it is a switch until the hardware has said. */ + uint32_t depth; + uint32_t twiddle3; + /* Bytes of border map ahead of level 0 - the texels the border modes + * read - and the colour it currently holds. Zero for none. */ + uint32_t border_map; + /* Bytes of guard either side of the map. The unit displaces its whole + * fetch when a border mode is selected - measured - and the guard is + * what that displacement lands in instead of in another object: a + * multiple of the map's size, sgx_border_guard_mult(), so that a + * displacement of more than one map is still inside this one. */ + uint32_t border_guard; + float border_rgba[4]; + int border_filled; + /* A format wider than the unit fetches in one go is stored as + * several planes - the DDK's chunks - each a whole texture of the + * plane's own format at chunk_size bytes from the last, and sampled + * separately (sgx535pixfmts.h: RGBA32F is four F32 chunks). One for + * everything else. format names the plane's format; the CPU sees + * the texel interleaved, through the staging copy. */ + uint32_t nchunks; + uint32_t chunk_size; + struct sgx_resource_level level[SGX_MAX_LEVELS]; + /* A twiddled level cannot be written a row at a time, so the CPU gets + * a linear staging copy and the swizzle happens on unmap. */ + /* A cached copy of a buffer's contents, for the CPU to read. + * + * The object itself is mapped write-combining, and a CPU read from + * write-combining memory is an order of magnitude slower than a cached + * one. The hardware transform path reads the caller's vertex data once + * per triangle corner per attribute - tens of thousands of small + * scattered reads a frame - and that was over half the process's CPU + * time on a scene bound by it. One bulk sequential read into this + * serves them all. Refreshed when the object is written. */ + void *shadow; + uint32_t shadow_size; + int shadow_stale; + void *stage; + uint32_t stage_size; + uint32_t stage_level; + uint32_t stage_face; + int stage_dirty; + /* The staging copy already holds this face and level untwiddled, so a + * reader can have it as it stands. Cleared whenever anything writes + * the resource. Without it a texture sampled on the CPU - a vertex + * texture fetch is the only way that happens - was untwiddled whole + * on every draw and twiddled whole back again on the unmap. */ + int stage_valid; + unsigned bind; + int bound; /* it has a GPU address */ + /* The object came from somewhere else - a display server's scanout + * buffer, say - so this resource may unbind it but must not close the + * handle, which belongs to whoever made it. */ + int foreign; + void *map; + unsigned map_count; /* nesting, so unmap matches map */ +}; + +enum sgx_resource_status { + SGX_RESOURCE_OK = 0, + SGX_RESOURCE_BAD_FORMAT, + SGX_RESOURCE_BAD_SIZE, + SGX_RESOURCE_BAD_BIND, /* the format cannot serve that use */ + SGX_RESOURCE_NOMEM, + SGX_RESOURCE_NOT_MAPPED +}; + +const char *sgx_resource_status_name(enum sgx_resource_status st); + +/* The stride the hardware will read for this width and format, which is not + * negotiable. Zero for a format with no fixed pixel size. */ +uint32_t sgx_resource_stride(enum sgx_format fmt, uint32_t width); +/* Bytes a depth surface of this size needs at a sample count (1 or 4); zero + * for a count the part cannot reach. The colour target does not grow with the + * sample count - the part downscales it on store - and depth does. */ +uint32_t sgx_resource_depth_bytes(uint32_t w, uint32_t h, unsigned samples); + +/* Create a 2D resource. Buffers - no format, height 1 - are allowed and are + * laid out linearly, because nothing reads them through a surface descriptor. */ +/* The multisampled form. samples is 1 or 4; anything else is refused. */ +enum sgx_resource_status +sgx_resource_create_ms(struct sgx_resource *r, struct sgx_winsys *ws, + enum sgx_format fmt, uint32_t w, uint32_t h, + unsigned bind, unsigned samples); + +enum sgx_resource_status +sgx_resource_create(struct sgx_resource *r, struct sgx_winsys *ws, + enum sgx_format fmt, uint32_t w, uint32_t h, unsigned bind); + +/* A texture with a mip chain, and optionally in the twiddled layout. The + * hardware derives every level's address from the base address and the top + * level's size alone, so the layout here is not a free choice - it is the + * vendor's, sgx_resource_level_texels(). nlevels 0 or 1 is a single level. */ +enum sgx_resource_status +sgx_resource_create_tex(struct sgx_resource *r, struct sgx_winsys *ws, + enum sgx_format fmt, uint32_t w, uint32_t h, + unsigned bind, uint32_t nlevels, int twiddled); + +/* The same, stored as nchunks planes of fmt. A block-compressed format + * (ETC1) is twiddled whatever is asked: it has no linear form. */ +enum sgx_resource_status +sgx_resource_create_planes(struct sgx_resource *r, struct sgx_winsys *ws, + enum sgx_format fmt, uint32_t w, uint32_t h, + unsigned bind, uint32_t nlevels, int twiddled, + uint32_t nchunks); + +/* Bytes a texel of the whole resource takes as the CPU sees it: every + * plane's, or a block's for a block format. */ +unsigned sgx_resource_texel_bytes(const struct sgx_resource *r); + +/* The same, for a cube map: six faces of the given chain in one object, laid + * out face-major at the stride the texture unit assumes. Square and power of + * two, which is what the unit's log2 size fields can describe. */ +enum sgx_resource_status +sgx_resource_create_cube(struct sgx_resource *r, struct sgx_winsys *ws, + enum sgx_format fmt, uint32_t size, + unsigned bind, uint32_t nlevels); + +/* A volume: depth slices of every level in one object, twiddled in all + * three axes, which is what the unit's 3D texture type describes (every + * axis a log2, so a power of two throughout). The chain halves every axis + * to a floor of one. */ +enum sgx_resource_status +sgx_resource_create_3d(struct sgx_resource *r, struct sgx_winsys *ws, + enum sgx_format fmt, uint32_t w, uint32_t h, uint32_t d, + unsigned bind, uint32_t nlevels); + +/* Where a face's level begins, in bytes from the object's start. */ +uint32_t sgx_resource_face_offset(const struct sgx_resource *r, + uint32_t face, uint32_t level); + +/* What GL_CLAMP_TO_BORDER is given on this part, from SGX_BORDER: + * + * unset, 0 clamp to edge, reported as an approximation - the default + * fixed CLAMPBDR (5) on a texture with no map. Measured 2026-09-03: + * the part returns no frame at all for it. Kept as the probe + * that said so + * 1 CLAMPBDRMEM (6), the map's base as the texture address + * 2 CLAMPBDRMEM (6), the texels' address + * bdr1 CLAMPBDR (5) with a map, the map's base as the address + * bdr2 CLAMPBDR (5) with a map, the texels' address + * + * Settled on the part, 2026-09-03, and the answer is that none of these is a + * border. A border mode displaces the address the unit reads by the DDK's + * face offset for the texture's size, 48 + 8n texels, at every size tried - + * so 1 is the arrangement it wants and it renders the texture correctly with + * level 0 laid at that offset. It then clamps to the edge: a ruled texture + * reads back column 0 of the sample's own row outside itself, and nothing is + * fetched from the map. The default stays the approximation, nothing is laid + * out for it, and these values remain as the probes that measured it - + * work/fix-border/README.md. */ +enum sgx_border_policy { + SGX_BORDER_EDGE, + SGX_BORDER_FIXED, + /* Everything from here lays a map out. */ + SGX_BORDER_MAP_BASE, + SGX_BORDER_MAP_TEXELS, + SGX_BORDER_BDR_BASE, + SGX_BORDER_BDR_TEXELS +}; +enum sgx_border_policy sgx_border_policy(void); + +/* Whether a square, twiddled, sampled texture is laid out with a border map + * in front of it: only the two map policies. CLAMPBDR reads no map, so + * SGX_BORDER=fixed leaves every texture's layout alone - which is what makes + * it the cheap experiment. */ +int sgx_resource_border_maps(void); +/* The guard is a ruler, not a colour: every texel in it encodes where in the + * object it sits, so a probe that reads one gives the displacement and its + * sign rather than only "not the texture". Red is the low byte of the + * 64-byte block index from the object's start, green the high byte, and blue + * says which side - 0x20 below the map, 0xe0 above the chain. Neither blue + * is any landmark colour's, so the decode cannot be confused with the + * texture, the border, or the clear. */ +#define SGX_BORDER_GUARD_BLOCK 64u +#define SGX_BORDER_GUARD_LOW_B 32u +#define SGX_BORDER_GUARD_HIGH_B 224u +/* The map's own ruler, SGX_BORDER_RULER. A flat border colour says a probe + * read the map; a ruler says which byte of it - and, with the texture ruled + * too, tells "the unit clamped to the edge" apart from "the unit read a + * border face that lands inside the texture", which one colour each cannot. + * Off by default: the map holds the border colour, which is what it is for. */ +#define SGX_BORDER_MAP_B 96u +int sgx_border_ruler(void); + +/* How many map-sizes of guard go either side, from SGX_BORDER_GUARD. None by + * default: the guard is a probe's scaffolding and a texture that ships must + * not carry map-sizes of ruler around. Set it - four is what measured the + * displacement - for any run that puts a new address in front of the unit, + * because without it a wrong one is an MMU fault rather than a colour. Zero + * to sixty-four. */ +unsigned sgx_border_guard_mult(void); +/* The 3D twiddle order volumes are laid out in: SGX_3D_LAYOUT=slices for + * the per-slice form, the three-axis interleave otherwise. */ +enum xpsb_twiddle3 sgx_resource_twiddle3_order(void); +/* Fill the border map with one colour, in the resource's own format, and the + * guard either side of it with SGX_BORDER_GUARD_RGBA so that a read from the + * end of the map the unit was not handed is a colour and not a fault. A + * no-op when the map already holds that colour. */ +enum sgx_resource_status +sgx_resource_fill_border(struct sgx_resource *r, struct sgx_winsys *ws, + const float rgba[4]); + +/* Map one level of one face. For a twiddled resource this hands back linear + * staging that sgx_resource_unmap_level() swizzles into place - one level at + * a time, so a second map before the unmap would alias the first. Face is 0 + * for anything but a cube. A volume's staging is the whole level, slices at + * stride * level height. */ +enum sgx_resource_status +sgx_resource_map_face_level(struct sgx_resource *r, struct sgx_winsys *ws, + uint32_t face, uint32_t level, void **out, + uint32_t *stride); +enum sgx_resource_status +sgx_resource_map_level(struct sgx_resource *r, struct sgx_winsys *ws, + uint32_t level, void **out, uint32_t *stride); +/* As above, for a caller that only reads: the staging copy is kept rather than + * twiddled back, and a second map of the same level does no work at all. */ +enum sgx_resource_status +sgx_resource_map_level_ro(struct sgx_resource *r, struct sgx_winsys *ws, + uint32_t level, void **out, uint32_t *stride); +enum sgx_resource_status +sgx_resource_unmap_level(struct sgx_resource *r); + +/* The texel offset of a level from the start of its face, by the vendor's + * rule for the layout - GetMipMapOffset for a twiddled chain, + * GetNPOTMipMapOffset's STRIDE case for a linear one. Level nlevels is the + * texel count of the whole chain. */ +uint32_t sgx_resource_level_texels(uint32_t w0, uint32_t h0, int twiddled, + uint32_t level); + +/* Whether a size can be twiddled at all: both dimensions a power of two. */ +/* Diagnostic over-allocation for sampled textures, in rows of the top + * level's stride; SGX_TEX_SLACK_ROWS, zero by default. */ +unsigned sgx_resource_slack_rows(void); + +int sgx_resource_can_twiddle(uint32_t w, uint32_t h); + +/* Lay the resource out linearly, keeping its contents. A resource that is + * rendered into cannot stay twiddled: the raster pass writes linearly. Call it + * the first time one is attached as a colour target. */ +enum sgx_resource_status +sgx_resource_make_linear(struct sgx_resource *r, struct sgx_winsys *ws); + +void sgx_resource_destroy(struct sgx_resource *r, struct sgx_winsys *ws); + +/* pipe_context::buffer_map. Nested maps are counted so an unmap does not pull + * the mapping out from under another user. */ +enum sgx_resource_status +sgx_resource_map(struct sgx_resource *r, struct sgx_winsys *ws, void **out); +/* The same, but without draining the part first. Only for a caller that has + * promised the memory is not what the part is reading - PIPE_MAP_UNSYNCHRONIZED + * - because the wait here is what keeps every upload of a client array behind + * the previous frame's render. */ +enum sgx_resource_status +sgx_resource_map_nowait(struct sgx_resource *r, struct sgx_winsys *ws, + void **out, int nowait); +enum sgx_resource_status +sgx_resource_unmap(struct sgx_resource *r); + +/* Which address window a resource belongs in, from what it is bound as. */ +int sgx_resource_window(unsigned bind); + +#endif diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_sampler.c mesa-26.2.2/src/gallium/drivers/sgx/sgx_sampler.c --- mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_sampler.c 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/sgx_sampler.c 2026-09-08 10:57:36.681112312 +0200 @@ -0,0 +1,246 @@ +/* Sampler views - see sgx_sampler.h. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#include "sgx_sampler.h" +#include "xpsb_frame.h" + +#include +#include + +const char *sgx_sampler_status_name(enum sgx_sampler_status st) +{ + switch (st) { + case SGX_SAMPLER_OK: return "ok"; + case SGX_SAMPLER_BAD_UNIT: return "no such texture unit"; + case SGX_SAMPLER_BAD_FORMAT: return "format is not samplable"; + case SGX_SAMPLER_BAD_SIZE: return "size out of range"; + case SGX_SAMPLER_BAD_STRIDE: return "stride is not the required one"; + case SGX_SAMPLER_UNBOUND: return "no view bound to that unit"; + case SGX_SAMPLER_NO_BORDER_MAP: return "a border mode needs a border map"; + } + return "?"; +} + +/* Gallium's wrap enumeration is not the hardware's. The index here is into + * xpsb_tex_state_mode()'s addr_code[], which holds the hardware codes: 0 + * REPEAT, 2 CLAMP, 7 OGLCLAMP, 1 FLIP, 3 FLIPCLAMP, 5 CLAMPBDR, + * 6 CLAMPBDRMEM, 4 REPEATBDRMEM. */ +static uint32_t wrap_hw(enum sgx_wrap w) +{ + switch (w) { + case SGX_WRAP_CLAMP_TO_EDGE: return XPSB_WRAP_CLAMP; + case SGX_WRAP_MIRROR_REPEAT: return XPSB_WRAP_MIRROR; + case SGX_WRAP_CLAMP: return XPSB_WRAP_CLAMPGL; + case SGX_WRAP_MIRROR_CLAMP: return 4; + case SGX_WRAP_CLAMP_TO_BORDER_FIXED: + case SGX_WRAP_CLAMP_TO_BORDER_FIXED_MAP: + return XPSB_WRAP_CLAMP_BORDER; + case SGX_WRAP_CLAMP_TO_BORDER: return XPSB_WRAP_CLAMP_BORDER_MAP; + case SGX_WRAP_REPEAT_BORDER: return XPSB_WRAP_REPEAT_BORDER_MAP; + case SGX_WRAP_REPEAT: + default: return XPSB_WRAP_REPEAT; + } +} + +int sgx_sampler_uses_border(const struct sgx_sampler_state *st) +{ + return st && (sgx_wrap_needs_map(st->wrap_s) || + sgx_wrap_needs_map(st->wrap_t) || + sgx_wrap_needs_map(st->wrap_r)); +} + +uint32_t sgx_sampler_view_base(const struct sgx_sampler_view *v, + const struct sgx_sampler_state *st) +{ + if (!v) + return 0; + if (v->border_map && !v->border_base && sgx_sampler_uses_border(st)) + return v->gpu_va - v->border_map; + return v->gpu_va; +} + +static uint32_t filter_hw(enum sgx_filter f) +{ + return f == SGX_FILTER_LINEAR ? 1u : 0u; +} + +/* The ratios the unit has are the powers of two up to sixteen, so anything + * between two of them takes the lower - a caller asking for 6x gets 4x rather + * than being refused, which is what GL expects of a value it did not have to + * clamp itself. Below 2x there is no anisotropy at all. */ +uint32_t sgx_aniso_code(float max_anisotropy) +{ + if (!(max_anisotropy >= 2.0f)) /* also catches NaN */ + return 0u; + if (max_anisotropy >= 16.0f) + return 4u; + if (max_anisotropy >= 8.0f) + return 3u; + if (max_anisotropy >= 4.0f) + return 2u; + return 1u; +} + +enum sgx_sampler_status sgx_sampler_view_check(const struct sgx_sampler_view *v) +{ + unsigned bpp; + + if (!v) + return SGX_SAMPLER_BAD_SIZE; + if (!sgx_format_supported(v->format, SGX_FMT_BIND_SAMPLER)) + return SGX_SAMPLER_BAD_FORMAT; + if (!v->width || !v->height || + v->width > (uint32_t)sgx_screen_cap(SGX_CAP_MAX_TEXTURE_SIZE) || + v->height > (uint32_t)sgx_screen_cap(SGX_CAP_MAX_TEXTURE_SIZE)) + return SGX_SAMPLER_BAD_SIZE; + + /* The stride is not free. The surface descriptor holds a width and a + * format, not a pitch, so the hardware computes where a row starts; + * a texture whose rows are packed differently is read as if they were + * not. xpsb_surface_check() is the same rule, applied here so the + * reason is reportable. */ + bpp = xpsb_format_bpp((uint32_t)v->format); + /* A block format has no texel size and only the twiddled form. */ + if (sgx_format_block(v->format, NULL, NULL, NULL)) { + if (!v->twiddled) + return SGX_SAMPLER_BAD_STRIDE; + } else if (!bpp) { + return SGX_SAMPLER_BAD_FORMAT; + } + /* A twiddled surface is addressed by log2 of its size and carries no + * pitch, so the rule above is not merely satisfied differently, it + * does not exist. Power of two is required instead. */ + if (v->twiddled) { + if ((v->width & (v->width - 1u)) || + (v->height & (v->height - 1u))) + return SGX_SAMPLER_BAD_SIZE; + } else if (v->stride != xpsb_linear_stride((uint32_t)v->format, + v->width)) { + return SGX_SAMPLER_BAD_STRIDE; + } + /* A volume is described in log2 on every axis, so it is twiddled and + * its depth is a power of two within the same size cap. */ + if (v->depth > 1) { + if (!v->twiddled || v->cube || + (v->depth & (v->depth - 1u)) || + v->depth > (uint32_t)sgx_screen_cap(SGX_CAP_MAX_TEXTURE_SIZE)) + return SGX_SAMPLER_BAD_SIZE; + } + /* The map's table is square: a map on any other shape describes + * texels at offsets the unit will not read from. */ + if (v->border_map && + (!v->twiddled || v->cube || v->depth > 1 || + v->width != v->height || + v->border_map != bpp * xpsb_border_map_face_offset(v->width))) + return SGX_SAMPLER_BAD_SIZE; + if (v->nlevels > 15u) + return SGX_SAMPLER_BAD_SIZE; + return SGX_SAMPLER_OK; +} + +enum sgx_sampler_status sgx_sampler_words(const struct sgx_sampler_view *v, + const struct sgx_sampler_state *st, + uint32_t *w0, uint32_t *w1) +{ + struct xpsb_surface_desc d; + enum sgx_sampler_status s; + + if (!v || !st || !w0 || !w1) + return SGX_SAMPLER_BAD_SIZE; + s = sgx_sampler_view_check(v); + if (s != SGX_SAMPLER_OK) + return s; + /* Not approximated here: which mode replaces it, and that it has, + * is the caller's to say. */ + if (sgx_sampler_uses_border(st) && !v->border_map) + return SGX_SAMPLER_NO_BORDER_MAP; + + memset(&d, 0, sizeof(d)); + d.offset = sgx_sampler_view_base(v, st); + d.w = v->width; + d.h = v->height; + d.stride = v->stride; + d.format = (uint32_t)v->format; + d.umode = wrap_hw(st->wrap_s); + d.vmode = wrap_hw(st->wrap_t); + d.border_map = v->border_map ? 1u : 0u; + d.gamma = v->srgb ? 1u : 0u; + d.chanrep = v->chanrep ? 1u : 0u; + d.twiddled = v->twiddled; + /* The slice axis. A volume is one of the log2 encodings, so it is + * twiddled whatever the resource said - and the assignment that used + * to follow this block undid that. */ + if (v->depth > 1) { + d.depth = v->depth; + d.smode = wrap_hw(st->wrap_r); + d.twiddled = 1u; + } + d.minfilter = filter_hw(st->min_filter); + d.magfilter = filter_hw(st->mag_filter); + /* The ratio alone does nothing: the unit only reads ANISOCTL when the + * filter field itself names ANISO. Programming the ratio against a + * LINEAR filter leaves the sampling isotropic and silently so, which + * is what this descriptor did until it was measured on hardware. */ + if (sgx_aniso_code(st->max_anisotropy)) { + if (d.minfilter == XPSB_FILTER_LINEAR) + d.minfilter = XPSB_FILTER_ANISO; + if (d.magfilter == XPSB_FILTER_LINEAR) + d.magfilter = XPSB_FILTER_ANISO; + } + /* A chain is only described when the sampler asks for one: a texture + * with levels but a non-mipmapping filter must still read as level 0 + * only, which is what the saturated field means. */ + /* A cube map is the twiddled encoding with a texture type of its own; + * the six faces are behind the one address the unit is given. */ + if (v->cube) { + d.twiddled = 1u; + d.textype = XPSB_TEXTYPE_CEM; + } + if (st->mip_filter != SGX_MIPFILTER_NONE && v->nlevels > 1) { + d.nlevels = v->nlevels; + d.mipfilter = st->mip_filter == SGX_MIPFILTER_LINEAR ? 1u : 0u; + /* A max-LOD below the chain's last level clamps the unit + * rather than shortening the chain, so the levels above stay + * addressable for a later sampler that wants them. Negative + * and enormous values are the GL defaults for "no clamp". */ + if (st->max_lod >= 0.0f && + st->max_lod < (float)(v->nlevels - 1u)) { + d.max_level = (uint32_t)st->max_lod; + d.max_level_set = 1u; + } + } else if ((v->cube && v->nlevels > 1) || v->levels_undescribed) { + /* A texture that has levels in memory but is sampled without a + * mipmap filter must still be described as mipmapped, clamped + * to level zero. The vendor does exactly this and avoids the + * not-mipmapped code for that case; saturating the field on a + * mipmapped cube is the shape it steps around. + * + * The same holds for a chain this view does not describe. Left + * as not mipmapped the unit still worked out an LOD and + * fetched against it, which is what washed out every minified + * sample in glmark2's terrain: its bloom pass reduces 1280x720 + * to 256x256, an LOD of about 2.3, and read past the one level + * it had been given. */ + d.nlevels = 2u; + d.max_level = 0u; + d.max_level_set = 1u; + d.mipfilter = 0u; + } + /* A linear chain - the STRIDE type, which is every non-power-of-two + * texture - samples half a level nearer than asked: the vendor adds + * a -0.5 LOD adjust to any NPOT texture whose minification filter + * mipmaps (opengles2/texmgmt.c:3420-3426, (-0.5 - DADJUST_MIN) * 8), + * on top of the sampler's own bias. A twiddled chain gets none. */ + d.lod_bias = xpsb_lod_bias_code(d.nlevels > 1 && !d.twiddled ? + st->lod_bias - 0.5f : st->lod_bias); + d.aniso = sgx_aniso_code(st->max_anisotropy); + + /* The captured mode, not the blob's: it is the word the working stream + * carries. */ + if (xpsb_tex_state_mode(w0, w1, &d, XPSB_TEXCTL_CAPTURE)) + return SGX_SAMPLER_BAD_FORMAT; + return SGX_SAMPLER_OK; +} diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_sampler.h mesa-26.2.2/src/gallium/drivers/sgx/sgx_sampler.h --- mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_sampler.h 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/sgx_sampler.h 2026-09-08 10:57:36.681120472 +0200 @@ -0,0 +1,223 @@ +/* Sampler views and sampler state: binding a texture to a unit. + * + * The two hardware words a texture unit needs are built by + * xpsb_tex_state_mode() in tools/xpsb-open, which is reverse engineered and + * exercised on hardware. This file is not a second implementation of that - + * it is the driver-side question of which textures are bound where, what a + * Gallium sampler state means in the hardware's terms, and what has to be + * refused because the descriptor cannot express it. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#ifndef _SGX_SAMPLER_H_ +#define _SGX_SAMPLER_H_ + +#include + +#include "sgx_screen.h" + +/* Only one sampled unit can actually be submitted - see the note on + * xpsb_frame_set_ntex(). More can be described, which is how the tables were + * checked, so the limit is on submitting rather than on binding. */ +#define SGX_MAX_SAMPLERS 8 +/* What the frame can actually carry. Not a hardware limit: the part has eight + * units and the vendor's own compiler fills a format record for each, in a + * loop over eight guarded by the fragment program's sampler-used mask + * (gl-re/textures.md section 2.2). + * + * What bounded it here was the primary PDS data segment: the captured slot + * holds sixteen dwords, and a descriptor triple per issue - control and format + * at bank 0 dwords 2 + 2i and 3 + 2i, address at bank 1 dword 2 + 2i - runs + * past that at the fourth. A list too wide for that slot is built in the spare + * block above the video constants now, which takes eight, and every slot is + * addressed rather than stepped, so the fourth no longer lands on the first. + * The frame still binds two units at fixed addresses; the rest are carried by + * each record's own PDS data, which names the texture's own base. */ +#define SGX_MAX_TEX_UNITS 8 + +/* Sampler units the vertex stage may declare. The part has no vertex + * texture unit at all - the draw module samples these on the CPU - so this + * is bounded by what that costs rather than by the hardware. Two is what + * glmark2's terrain wants (a displacement map and a normal map). */ +#define SGX_MAX_VTX_SAMPLERS 4 +#define SGX_MAX_SUBMITTABLE_SAMPLERS 1 + +/* pipe_sampler_state, reduced to what the hardware has a field for. */ +/* The unit's address modes, named as the DDK names them + * (EURASIA_PDS_DOUTT0_ADDRMODE_*). Its own numbering is 0 REPEAT, 1 FLIP, + * 2 CLAMP, 3 FLIPCLAMP, 4-6 the border-map forms, 7 OGLCLAMP; this enum is + * the driver's own and xpsb_tex_state_mode() maps it. + * + * A border colour on this part is not a register but a border map laid out + * in front of the texture in memory, in the texture's own format, at the + * offsets the DDK's EURASIA_TAG_BORDERMAP_* table gives (sgxdefs.h:5061). + * Nothing in the vendor stack ever builds one, so the modes that read it are + * offered only on a view that carries a map - sgx_sampler_view.border_map - + * and refused otherwise, which the caller reports rather than approximates. */ +enum sgx_wrap { + SGX_WRAP_REPEAT = 0, + SGX_WRAP_CLAMP_TO_EDGE, + SGX_WRAP_MIRROR_REPEAT, + /* Legacy GL_CLAMP: clamps to the edge of the border, which without a + * border map is a half-texel differently placed than clamp-to-edge. + * The hardware has it outright - it is the mode Xpsb.so's own + * clampGL symbol selects. */ + SGX_WRAP_CLAMP, + /* GL_MIRROR_CLAMP_TO_EDGE. */ + SGX_WRAP_MIRROR_CLAMP, + /* GL_CLAMP_TO_BORDER: EURASIA_PDS_DOUTT0_ADDRMODE_CLAMPBDRMEM (6). + * Measured 2026-09-03: it displaces the address by the DDK's face + * offset and then clamps to the edge. No border texel is read, from + * the map or anywhere else, so this is a probe and not a mode - the + * driver's own clamp-to-edge is the same picture for nothing. */ + SGX_WRAP_CLAMP_TO_BORDER, + /* CLAMPBDR (5), the form without the MEM suffix, on a texture that + * has no map. Measured 2026-09-03: the part renders no frame at all + * for it - not the draw and not even the clear - which is what code 7 + * does too. Kept because it is the probe that produced that, and + * sgx_pipe.c selects it only under SGX_BORDER=fixed and says so. */ + SGX_WRAP_CLAMP_TO_BORDER_FIXED, + /* CLAMPBDR (5) again, on a texture that does have one. Measured + * 2026-09-03: indistinguishable from CLAMPBDRMEM at both ends of the + * map - the two codes are not told apart by whether a map is there, + * and neither reads one. Kept because the pair is what says that. */ + SGX_WRAP_CLAMP_TO_BORDER_FIXED_MAP, + /* REPEATBDRMEM (4): repeat, with the border map's texels at the + * seams. No GL wrap mode is this; it is reachable for the same + * experiment. */ + SGX_WRAP_REPEAT_BORDER +}; + +/* Whether a mode reads the border map, and so needs a view that has one. */ +static inline int sgx_wrap_needs_map(enum sgx_wrap w) +{ + return w == SGX_WRAP_CLAMP_TO_BORDER || w == SGX_WRAP_REPEAT_BORDER || + w == SGX_WRAP_CLAMP_TO_BORDER_FIXED_MAP; +} + +enum sgx_filter { + SGX_FILTER_NEAREST = 0, + SGX_FILTER_LINEAR +}; + +/* How the two levels either side of the computed LOD are combined. NONE is + * not a filter but the absence of a chain: it makes the descriptor say "not + * mipmapped" however many levels the texture happens to carry. */ +enum sgx_mipfilter { + SGX_MIPFILTER_NONE = 0, + SGX_MIPFILTER_NEAREST, + SGX_MIPFILTER_LINEAR +}; + +struct sgx_sampler_state { + /* wrap_r is the slice axis of a volume - DOUTT0 SADDRMODE - and is + * not written for a 2D view. */ + enum sgx_wrap wrap_s, wrap_t, wrap_r; + enum sgx_filter min_filter, mag_filter; + enum sgx_mipfilter mip_filter; + /* GL_TEXTURE_LOD_BIAS, in levels. The unit has a six-bit adjust for + * it - see xpsb_lod_bias_code() - so the range is what + * PIPE_CAP_MAX_TEXTURE_LOD_BIAS reports rather than unbounded. */ + float lod_bias; + /* GL_TEXTURE_MAX_LOD as a level index. There is no minimum-LOD field + * on this part (the DDK gates one behind SGX_FEATURE_TAG_MINLOD, which + * is a later core), so min_lod cannot be honoured and the caller's is + * not carried here at all. */ + float max_lod; + /* GL_TEXTURE_MAX_ANISOTROPY_EXT, as the caller asked for it. Rounded + * down to a ratio the unit has by sgx_aniso_code(). */ + float max_anisotropy; + /* GL_TEXTURE_BORDER_COLOR, rgba. Not a descriptor field: it is what + * the view's border map is filled with when a border mode is bound - + * sgx_resource_fill_border() - so it is carried here for that. */ + float border_color[4]; + /* GL_TEXTURE_COMPARE_MODE: zero for none, else the pipe comparison + * plus one. The unit has no comparison on this core, so the pipe + * bakes it into the fragment program from this. */ + unsigned compare; +}; + +/* The anisotropic ratio field's encoding: 0 none, 1 2x, 2 4x, 3 8x, 4 16x. + * A ratio the hardware does not have rounds down, which is what GL asks an + * implementation to do with a value between two it supports. */ +#define SGX_MAX_ANISOTROPY 16.0f +uint32_t sgx_aniso_code(float max_anisotropy); + +/* pipe_sampler_view: the texture itself. */ +struct sgx_sampler_view { + uint32_t gpu_va; + uint32_t width, height, stride; + enum sgx_format format; + /* The chain the view exposes. The hardware has no first-level field - + * gl-re/textures.md section 3.1 - so a view that does not start at + * level 0 has to point gpu_va at its own base level instead. */ + uint32_t nlevels; + /* The resource holds a mip chain that this view does not describe - a + * linear chain without SGX_LINEAR_MIPS. The unit still minifies, so + * such a view has to be described as mipmapped and clamped to level + * zero rather than as not mipmapped; see sgx_sampler_words(). */ + uint32_t levels_undescribed; + uint32_t twiddled; + /* A cube map: six faces in one object, which the unit reaches from + * this one address. The faces must be square and a power of two - + * their size is described as log2 - and the layout is the resource + * layer's sgx_resource_create_cube(). */ + uint32_t cube; + /* A volume: slices, a power of two, twiddled in the order + * sgx_resource_create_3d() chose. 0 and 1 are a 2D view. */ + uint32_t depth; + /* Bytes of border map in front of gpu_va, zero for none. A border + * mode names the map's base as the texture's address - + * sgx_sampler_view_base() - and every other mode names gpu_va; + * border_base non-zero names gpu_va for a border mode as well, the + * other reading of "offset to the first map face". */ + uint32_t border_map, border_base; + /* The texels are sRGB and the unit is to decode them before it + * filters: DOUTT0 GAMMA, bit 27 (sgxdefs.h:4344). The DDK calls the + * feature "gamma correction on texture reads" + * (sgxfeaturedefs.h:44-46), which is upstream of the filter and so + * the order GL asks for; set only for the formats the unit unpacks + * as four unsigned bytes - sgx_srgb_in_unit(). */ + uint32_t srgb; + /* DOUTT0 CHANREPLICATE on a format the one-byte rule leaves it off + * for - the AL88 experiment, sgx_pipe.c. */ + uint32_t chanrep; + /* Planes the texel is stored in and the bytes between them - the + * resource's chunks. One and zero for a texel stored whole. */ + uint32_t nchunks, chunk_size; +}; + +enum sgx_sampler_status { + SGX_SAMPLER_OK = 0, + SGX_SAMPLER_BAD_UNIT, + SGX_SAMPLER_BAD_FORMAT, /* not samplable on this hardware */ + SGX_SAMPLER_BAD_SIZE, + SGX_SAMPLER_BAD_STRIDE, /* not bpp * align(width, 32) */ + SGX_SAMPLER_UNBOUND, /* no view on that unit */ + SGX_SAMPLER_NO_BORDER_MAP /* a border mode on a view without one */ +}; + +const char *sgx_sampler_status_name(enum sgx_sampler_status st); + +/* Validate a view against what the hardware can address. Called before the + * words are built, because xpsb_tex_state() reports a bad surface the same way + * whatever is wrong with it. */ +enum sgx_sampler_status sgx_sampler_view_check(const struct sgx_sampler_view *v); + +/* Build the two texture-unit words for a view and a sampler state. */ +enum sgx_sampler_status sgx_sampler_words(const struct sgx_sampler_view *v, + const struct sgx_sampler_state *st, + uint32_t *w0, uint32_t *w1); + +/* The address the descriptor's third word names for this pair: gpu_va, or + * the border map's base in front of it when the state reads the map. Which + * of the two the unit wants for a map is the first thing the hardware is + * asked - SGX_BORDER=2 in sgx_pipe.c keeps gpu_va either way. */ +uint32_t sgx_sampler_view_base(const struct sgx_sampler_view *v, + const struct sgx_sampler_state *st); +/* Whether the state selects a mode that reads the border map on any axis. */ +int sgx_sampler_uses_border(const struct sgx_sampler_state *st); + +#endif diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_screen.c mesa-26.2.2/src/gallium/drivers/sgx/sgx_screen.c --- mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_screen.c 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/sgx_screen.c 2026-09-08 10:57:36.681129443 +0200 @@ -0,0 +1,248 @@ +/* The driver-side screen - see sgx_screen.h. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#include "sgx_screen.h" +#include "sgx_sampler.h" +#include "usse_ir.h" +#include "xpsb_frame.h" + +/* The vertex primary-attribute bank is 32 dwords per instance, measured by + * walking a DOUTD past the end until the frame changed. Four dwords per + * attribute, and the transform uniforms have to fit in the same bank, which is + * why the uniform count is what is left rather than a separate limit. */ +#define SGX_PA_BANK_DWORDS 32 +#define SGX_UNIFORM_DWORDS 18 /* the lit vertex program's block */ + +static const int sgx_caps[SGX_CAP_COUNT] = { + /* EURASIA_RENDERSIZE_MAXX and MAXY. sgxdefs.h:389-425 gives 8192 for + * SGX545, 4096 for the SGX543 family and 2048 for everything else - + * which is the arm this part takes - and SGXAddRenderTarget() refuses + * anything larger outright (sgxrender_targets.c:701-708). This said + * 4096, which was the scene cookie's twelve-bit extent field rather + * than a rendering limit: the cookie could describe a target the ISP + * will not render. The kernel already refused past 2048 + * (psb_sgx_render.c, the submit's size check), so the over-claim was + * a failed allocation somewhere above rather than a bad picture - + * but it was still a claim the driver could not honour. */ + [SGX_CAP_MAX_RENDER_WIDTH] = XPSB_RENDER_SIZE_MAX, + [SGX_CAP_MAX_RENDER_HEIGHT] = XPSB_RENDER_SIZE_MAX, + [SGX_CAP_MAX_TEXTURE_SIZE] = 2048, + /* Two: the context has a buffer for each and sgx_use_texture() + * refuses a third. Advertising eight let a caller bind views the + * driver then dropped without a word. */ + /* What the GL frontend needs, not what the frame can issue. The + * hardware path serves SGX_MAX_TEX_UNITS and sgx_use_texture() + * refuses a view above that with a diagnostic - but reporting the + * smaller number here drops the GL version Mesa exposes far enough + * that glamor picks a path this driver has no format for, and the X + * server does not start at all. */ + [SGX_CAP_MAX_TEXTURE_UNITS] = 8, + [SGX_CAP_MAX_VERTEX_ATTRIBS] = + (SGX_PA_BANK_DWORDS - SGX_UNIFORM_DWORDS) / 4, + [SGX_CAP_MAX_VERTEX_UNIFORMS] = SGX_UNIFORM_DWORDS, + [SGX_CAP_MAX_RENDER_TARGETS] = 1, + [SGX_CAP_NPOT_TEXTURES] = 1, + [SGX_CAP_OCCLUSION_QUERY] = 1, /* VISTEST_RESULT, eight dwords */ + /* Four, as 2x2. SGXAddRenderTarget() takes 1x1, 2x1, 1x2 and 2x2 + * (sgxrender_targets.c:738-753), but the two-sample forms need + * SGX_FEATURE_MSAA_2X_IN_X or _IN_Y and the SGX535 block of + * sgxfeaturedefs.h (189-232) defines neither, so two samples a pixel + * is not reachable on this part and is refused rather than rounded. */ + [SGX_CAP_MAX_SAMPLES] = XPSB_MSAA_4X, + [SGX_CAP_TILE_SIZE] = 16, + [SGX_CAP_CORE_CLOCK_MHZ] = 200, +}; + +static const char *const sgx_cap_names[SGX_CAP_COUNT] = { + [SGX_CAP_MAX_RENDER_WIDTH] = "max render width", + [SGX_CAP_MAX_RENDER_HEIGHT] = "max render height", + [SGX_CAP_MAX_TEXTURE_SIZE] = "max texture size", + [SGX_CAP_MAX_TEXTURE_UNITS] = "max texture units", + [SGX_CAP_MAX_VERTEX_ATTRIBS] = "max vertex attributes", + [SGX_CAP_MAX_VERTEX_UNIFORMS] = "max vertex uniforms", + [SGX_CAP_MAX_RENDER_TARGETS] = "max render targets", + [SGX_CAP_NPOT_TEXTURES] = "npot textures", + [SGX_CAP_OCCLUSION_QUERY] = "occlusion query", + [SGX_CAP_MAX_SAMPLES] = "max samples", + [SGX_CAP_TILE_SIZE] = "tile size", + [SGX_CAP_CORE_CLOCK_MHZ] = "core clock MHz", +}; + +int sgx_screen_cap(enum sgx_cap cap) +{ + if (cap < 0 || cap >= SGX_CAP_COUNT) + return 0; + return sgx_caps[cap]; +} + +const char *sgx_screen_cap_name(enum sgx_cap cap) +{ + if (cap < 0 || cap >= SGX_CAP_COUNT) + return "?"; + return sgx_cap_names[cap]; +} + +static const struct { + enum sgx_format fmt; + unsigned bind; + unsigned bytes; +} sgx_formats[] = { + /* Rendered to as well as sampled. A GLES 2 context will not take an + * alpha texture as a colour attachment, but it takes a red one, and + * this code is the single 8-bit channel that serves both - so the X + * server gets the depth-eight target its glyph atlas is built in. + * Without it glamor found its test of one incomplete, fell back, and + * composited no glyphs at all. */ + { SGX_FMT_A8, SGX_FMT_BIND_SAMPLER | SGX_FMT_BIND_RENDER, 1 }, + { SGX_FMT_AL88, SGX_FMT_BIND_SAMPLER | SGX_FMT_BIND_RENDER, 2 }, + { SGX_FMT_A4R4G4B4, SGX_FMT_BIND_SAMPLER, 2 }, + { SGX_FMT_A1R5G5B5, SGX_FMT_BIND_SAMPLER | SGX_FMT_BIND_RENDER | + SGX_FMT_BIND_DISPLAY, 2 }, + { SGX_FMT_R5G6B5, SGX_FMT_BIND_SAMPLER | SGX_FMT_BIND_RENDER | + SGX_FMT_BIND_DISPLAY, 2 }, + { SGX_FMT_A8R8G8B8, SGX_FMT_BIND_SAMPLER | SGX_FMT_BIND_RENDER | + SGX_FMT_BIND_DISPLAY, 4 }, + { SGX_FMT_A8B8G8R8, SGX_FMT_BIND_SAMPLER | SGX_FMT_BIND_RENDER | + SGX_FMT_BIND_DISPLAY, 4 }, + /* Sampler only. The render target's pixel-format field has no encoding + * for a subsampled format, so a driver that offers these as render + * targets is offering something the descriptor cannot express. */ + { SGX_FMT_YUY2, SGX_FMT_BIND_SAMPLER, 2 }, + { SGX_FMT_UYVY, SGX_FMT_BIND_SAMPLER, 2 }, + /* One plane's texel: a wider format is several of these. */ + { SGX_FMT_U16, SGX_FMT_BIND_SAMPLER, 2 }, + { SGX_FMT_S16, SGX_FMT_BIND_SAMPLER, 2 }, + { SGX_FMT_F16, SGX_FMT_BIND_SAMPLER, 2 }, + { SGX_FMT_U1616, SGX_FMT_BIND_SAMPLER, 4 }, + { SGX_FMT_S1616, SGX_FMT_BIND_SAMPLER, 4 }, + { SGX_FMT_F1616, SGX_FMT_BIND_SAMPLER, 4 }, + { SGX_FMT_F32, SGX_FMT_BIND_SAMPLER, 4 }, + /* Eight bytes a 4x4 block; no fixed size a texel. */ + { SGX_FMT_ETC1, SGX_FMT_BIND_SAMPLER, 0 }, +}; + +static const unsigned sgx_format_n = + sizeof sgx_formats / sizeof sgx_formats[0]; + +int sgx_format_supported(enum sgx_format fmt, unsigned bind_mask) +{ + unsigned i; + + if (!bind_mask) + return 0; + for (i = 0; i < sgx_format_n; i++) + if (sgx_formats[i].fmt == fmt) + return (sgx_formats[i].bind & bind_mask) == bind_mask; + return 0; +} + +unsigned sgx_format_bytes(enum sgx_format fmt) +{ + unsigned i; + + for (i = 0; i < sgx_format_n; i++) + if (sgx_formats[i].fmt == fmt) + return sgx_formats[i].bytes; + return 0; +} + +unsigned sgx_format_block(enum sgx_format fmt, unsigned *bw, unsigned *bh, + unsigned *bytes) +{ + if (fmt != SGX_FMT_ETC1) + return 0; + if (bw) *bw = 4; + if (bh) *bh = 4; + if (bytes) *bytes = 8; + return 1; +} + +unsigned sgx_format_texclass(enum sgx_format fmt) +{ + switch (fmt) { + case SGX_FMT_U16: return UIR_TEXCLASS_U16; + case SGX_FMT_S16: return UIR_TEXCLASS_S16; + case SGX_FMT_F16: return UIR_TEXCLASS_F16; + case SGX_FMT_U1616: return UIR_TEXCLASS_U1616; + case SGX_FMT_S1616: return UIR_TEXCLASS_S1616; + case SGX_FMT_F1616: return UIR_TEXCLASS_F1616; + case SGX_FMT_F32: return UIR_TEXCLASS_F32; + default: return UIR_TEXCLASS_U8888; + } +} + +/* XPSB_SURF_FMT() in xpsb_frame.h: the format sits at [31:24] under a fixed + * 0x60 tag, which is why an unsupported format cannot simply be passed + * through - it would produce a valid-looking word for a format the ISP does + * not decode. */ +uint32_t sgx_surface_format_word(enum sgx_format fmt) +{ + if (!sgx_format_supported(fmt, SGX_FMT_BIND_RENDER)) + return 0; + return 0x60000000u | ((uint32_t)fmt << 24); +} + +static unsigned pack_chan(float v, unsigned bits) +{ + unsigned top = (1u << bits) - 1u; + + if (!(v > 0.0f)) /* also NaN */ + return 0; + if (v >= 1.0f) + return top; + return (unsigned)(v * (float)top + 0.5f); +} + +/* The byte orders are gl-re/textures.md section 10.1, measured: 0xAARRGGBB + * for A8R8G8B8, 0xAABBGGRR for A8B8G8R8, and the 16-bit forms as their + * names read from the top bit down. AL88 is luminance in the low byte. */ +unsigned sgx_format_pack(enum sgx_format fmt, const float rgba[4], + unsigned char *out) +{ + uint32_t v; + unsigned n; + + switch (fmt) { + case SGX_FMT_A8: + v = pack_chan(rgba[3], 8); + n = 1; + break; + case SGX_FMT_AL88: + v = pack_chan(rgba[0], 8) | (pack_chan(rgba[3], 8) << 8); + n = 2; + break; + case SGX_FMT_A4R4G4B4: + v = (pack_chan(rgba[3], 4) << 12) | (pack_chan(rgba[0], 4) << 8) | + (pack_chan(rgba[1], 4) << 4) | pack_chan(rgba[2], 4); + n = 2; + break; + case SGX_FMT_A1R5G5B5: + v = (pack_chan(rgba[3], 1) << 15) | (pack_chan(rgba[0], 5) << 10) | + (pack_chan(rgba[1], 5) << 5) | pack_chan(rgba[2], 5); + n = 2; + break; + case SGX_FMT_R5G6B5: + v = (pack_chan(rgba[0], 5) << 11) | (pack_chan(rgba[1], 6) << 5) | + pack_chan(rgba[2], 5); + n = 2; + break; + case SGX_FMT_A8R8G8B8: + v = (pack_chan(rgba[3], 8) << 24) | (pack_chan(rgba[0], 8) << 16) | + (pack_chan(rgba[1], 8) << 8) | pack_chan(rgba[2], 8); + n = 4; + break; + case SGX_FMT_A8B8G8R8: + v = (pack_chan(rgba[3], 8) << 24) | (pack_chan(rgba[2], 8) << 16) | + (pack_chan(rgba[1], 8) << 8) | pack_chan(rgba[0], 8); + n = 4; + break; + default: + return 0; + } + for (unsigned i = 0; i < n; i++) + out[i] = (unsigned char)(v >> (8u * i)); + return n; +} diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_screen.h mesa-26.2.2/src/gallium/drivers/sgx/sgx_screen.h --- mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_screen.h 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/sgx_screen.h 2026-09-08 10:57:36.681147943 +0200 @@ -0,0 +1,119 @@ +/* The driver-side screen: what the hardware can do, in plain structs. + * + * A pipe_screen answers two kinds of question - what are the limits, and is + * this format supported - and both are answers this project measured rather + * than read from a datasheet. That is what makes this worth writing before the + * Mesa glue exists: the numbers are the reverse engineering, and getting one + * wrong is a driver that reports a capability it does not have. + * + * Every limit below carries where it came from. "measured" means a hardware + * experiment in work/; "encoding" means a field width that bounds it whatever + * the silicon would otherwise allow. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#ifndef _SGX_SCREEN_H_ +#define _SGX_SCREEN_H_ + +#include + +/* Formats, in the hardware's own numbering - XPSB_FMT_* in xpsb_frame.h. A + * driver's format table is this set and nothing else; anything absent has to + * be refused rather than guessed at. */ +enum sgx_format { + SGX_FMT_NONE = -1, + SGX_FMT_A8 = 0x00, + /* EURASIA_PDS_DOUTT1_TEXFORMAT_A4R4G4B4, sgxdefs.h:4579 - the vendor's + * GLES2 TexFormatARGB4444. Sampled only; the render-target word for it + * has not been seen on hardware. */ + SGX_FMT_A4R4G4B4 = 0x02, + /* Two interleaved 8-bit channels, gl-re/textures.md's al88. It is + * here for GL_EXT_texture_rg, which mesa gives a GLES 2 context only + * when both the one- and the two-channel format are renderable - and + * without that extension the X server has no depth-eight target for + * its glyph atlas. */ + SGX_FMT_AL88 = 0x07, + SGX_FMT_A1R5G5B5 = 0x04, + SGX_FMT_R5G6B5 = 0x05, + /* The 16-bit and float texel codes, EURASIA_PDS_DOUTT1_TEXFORMAT_* + * (sgxdefs.h:4586-4598): one channel a plane, or two for the 1616 + * forms. The vendor's R16F, R16_UNORM, R16_SNORM, G16R16F, G16R16 + * and R32F rows (sgx535pixfmts.h:1766-1830, 1670-1734, 1510) are + * one chunk each; RGBA16F, RGBA16 and RGBA32F are two, two and four + * chunks of the same codes (:1574, :1638, :1302), sampled a plane + * at a time. D32F, the vendor's depth texture, is the F32 code over + * the depth surface (:1478, TexFormatFloatDepth). Sampled only: the + * pixel back end's word for them is the pass-through PT1, which + * this driver's output packing does not produce. */ + SGX_FMT_U16 = 0x09, + SGX_FMT_S16 = 0x0a, + SGX_FMT_F16 = 0x0b, + SGX_FMT_A8R8G8B8 = 0x0c, + SGX_FMT_A8B8G8R8 = 0x0d, + SGX_FMT_U1616 = 0x0f, + SGX_FMT_S1616 = 0x10, + SGX_FMT_F1616 = 0x11, + SGX_FMT_F32 = 0x12, + /* ETC1, the DDK's PVRTIII (sgxdefs.h:4639, the non-543 branch); + * the vendor's TexFormatETC1RGB over PVRSRV_PIXEL_FORMAT_PVRTCIII + * (opengles2/texformat.c:255). Eight bytes a 4x4 block, twiddled at + * block granularity, and the TAG expands it to a packed 8888 texel + * (usp_sample.c:4736 unpacks it as B8G8R8A8). */ + SGX_FMT_ETC1 = 0x1b, + SGX_FMT_YUY2 = 0x1c, + SGX_FMT_UYVY = 0x1d, +}; + +/* What a format may be used for. YUV is a sampler-only format here: the + * pixel-format field of a render target has no encoding for it. */ +#define SGX_FMT_BIND_SAMPLER (1u << 0) +#define SGX_FMT_BIND_RENDER (1u << 1) +#define SGX_FMT_BIND_DISPLAY (1u << 2) + +/* pipe_screen::get_param, the subset with a measured answer. */ +enum sgx_cap { + SGX_CAP_MAX_RENDER_WIDTH, /* EURASIA_RENDERSIZE_MAXX */ + SGX_CAP_MAX_RENDER_HEIGHT, /* EURASIA_RENDERSIZE_MAXY */ + SGX_CAP_MAX_TEXTURE_SIZE, + SGX_CAP_MAX_TEXTURE_UNITS, + SGX_CAP_MAX_VERTEX_ATTRIBS, /* measured: the 32-dword PA bank */ + SGX_CAP_MAX_VERTEX_UNIFORMS, /* measured: what is left of that bank */ + SGX_CAP_MAX_RENDER_TARGETS, + SGX_CAP_NPOT_TEXTURES, + SGX_CAP_OCCLUSION_QUERY, + SGX_CAP_MAX_SAMPLES, /* 2x2, and nothing between it and one */ + SGX_CAP_TILE_SIZE, /* measured: the extent controls step in 16s */ + SGX_CAP_CORE_CLOCK_MHZ, /* measured: off the message bus */ + SGX_CAP_COUNT +}; + +int sgx_screen_cap(enum sgx_cap cap); +const char *sgx_screen_cap_name(enum sgx_cap cap); + +/* Nonzero if the format can be used for every binding in mask. */ +int sgx_format_supported(enum sgx_format fmt, unsigned bind_mask); + +/* Bytes per pixel, or 0 for a format that has no fixed size. */ +unsigned sgx_format_bytes(enum sgx_format fmt); + +/* A block-compressed format's block: texels a side and bytes a block. + * Returns 0 for a format stored a texel at a time. */ +unsigned sgx_format_block(enum sgx_format fmt, unsigned *bw, unsigned *bh, + unsigned *bytes); + +/* What the fetch returns for a format, as the compiler's UIR_TEXCLASS_*. */ +unsigned sgx_format_texclass(enum sgx_format fmt); + +/* The surface-descriptor word a render target of this format needs. */ +uint32_t sgx_surface_format_word(enum sgx_format fmt); + +/* One texel of the format from an rgba colour in [0,1], written little-endian + * into out. Returns the texel's size in bytes, or 0 for a format with no + * per-texel encoding (the packed YUV pair). What the border map is filled + * with, so it packs exactly what the unit unpacks. */ +unsigned sgx_format_pack(enum sgx_format fmt, const float rgba[4], + unsigned char *out); + +#endif diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_shader.c mesa-26.2.2/src/gallium/drivers/sgx/sgx_shader.c --- mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_shader.c 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/sgx_shader.c 2026-09-08 10:57:36.681186086 +0200 @@ -0,0 +1,3039 @@ +/* Shader objects - see sgx_shader.h. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#include "util/simple_mtx.h" + +#include "sgx_shader.h" +#include "sgx_context.h" +#include "xpsb_frame.h" +#include "sgx_hwtcl.h" +#include "usse.h" + +#include +#include +#include +#include + +const char *sgx_shader_status_name(enum sgx_shader_status st) +{ + switch (st) { + case SGX_SHADER_OK: return "ok"; + case SGX_SHADER_TRANSLATE_FAILED: return "TGSI did not translate"; + case SGX_SHADER_CODEGEN_FAILED: return "codegen failed"; + case SGX_SHADER_TOO_BIG: return "program too long"; + case SGX_SHADER_NO_END: return "program has no end"; + } + return "?"; +} + +/* Does the last instruction carry .end? + * + * This is checked rather than assumed. A USSE program whose last instruction + * lacks .end runs on into whatever follows it in memory, and this project has + * lost a machine to that. The backend sets it, but a shader that reached here + * without it is one the driver must refuse rather than submit. */ +static int ends(const uint64_t *code, unsigned n) +{ + struct usse_insn in; + char text[256]; + + if (!n) + return 0; + in.w0 = (uint32_t)(code[n - 1] & 0xffffffffu); + in.w1 = (uint32_t)(code[n - 1] >> 32); + text[0] = 0; + if (usse_disasm(&in, text, sizeof text)) + return 0; + return strstr(text, ".end") != NULL; +} + +/* TGSI semantic names, transcribed rather than included: this file compiles + * without a Mesa tree, which is what lets the compiler be tested on its own. */ +#define SGX_SEM_POSITION 0 +#define SGX_SEM_BCOLOR 2 +#define SGX_SEM_FOG 3 +#define SGX_SEM_PSIZE 4 +#define SGX_SEM_GENERIC 5 +#define SGX_SEM_FACE TGSI_SEM_FACE +#define SGX_SEM_TEXCOORD 19 + +/* Where the frame puts a varying of this semantic, or -1 when no pairing has + * been settled; see sgx_shader_vtx_layout(). */ +static int sgx_vtx_slot_of(unsigned sem, unsigned idx); +static unsigned sgx_vtx_nmap; + +/* Whether a vertex output has nowhere to go and its writes are dropped. */ +static int sgx_vs_out_dropped(const struct tgsi_io *io, unsigned i) +{ + switch (io->out_semantic[i]) { + case SGX_SEM_PSIZE: + case SGX_SEM_BCOLOR: + return 1; + case SGX_SEM_FOG: + return sgx_vtx_nmap && + sgx_vtx_slot_of(SGX_SEM_FOG, io->out_index[i]) < 0; + default: + return 0; + } +} + +/* Redirect every write of a dropped output into a virtual register nothing + * reads, so the program neither emits it nor widens the record for it. */ +static int sgx_vs_drop_outputs(struct uir_shader *s, const struct tgsi_io *io) +{ + size_t k; + unsigned i; + uint32_t dead[TGSI_PARSE_MAX_IO]; + + for (i = 0; i < io->nout && i < TGSI_PARSE_MAX_IO; i++) + dead[i] = sgx_vs_out_dropped(io, i) ? uir_alloc_virt(s) : ~0u; + for (k = 0; k < s->ninsns; k++) { + struct uir_ref *d = &s->insns[k].dst; + + if (d->cls == UIR_REG_OUT && d->index < io->nout && + d->index < TGSI_PARSE_MAX_IO && dead[d->index] != ~0u) { + d->cls = UIR_REG_VIRT; + d->index = dead[d->index]; + } + } + return 0; +} + +/* Which record slot each vertex output belongs in: position, colour, then the + * coordinate sets. Returns 0 if an output has nowhere to go. */ +static int sgx_vs_out_slots(const struct tgsi_io *io, unsigned char *slot, + unsigned nout) +{ + unsigned next_tc = 2, i; + int colour_used = 0; + + for (i = 0; i < nout && i < 16; i++) + if (io->out_semantic[i] == SGX_SEM_COLOR) + colour_used = 1; + + if (getenv("SGX_DUMP_VSOUT")) { + fprintf(sgx_log(), "sgx: vertex outputs:"); + for (i = 0; i < nout && i < 16; i++) + fprintf(sgx_log(), " [%u]=sem%u", i, + io->out_semantic[i]); + fprintf(sgx_log(), "\n"); + } + + for (i = 0; i < nout && i < 16; i++) { + switch (io->out_semantic[i]) { + case SGX_SEM_POSITION: + slot[i] = 0; + break; + case SGX_SEM_COLOR: + { + int want = sgx_vtx_slot_of(SGX_SEM_COLOR, + io->out_index[i]); + + if (want >= 0) { + slot[i] = (unsigned char)want; + break; + } + } + slot[i] = 1; + break; + /* Dropped: a point never takes the part's transform + * (sgx_hwtcl_prim_ok), and two-sided lighting is the draw + * module's (sgx_hwtcl_can), so nothing reads these. The + * writes go to a dead register, sgx_vs_drop_outputs(). */ + case SGX_SEM_PSIZE: + case SGX_SEM_BCOLOR: + slot[i] = 0; + break; + case SGX_SEM_FOG: + /* gl_FogFragCoord: an ordinary varying where the pair + * reads it, dropped where the pairing is settled and + * does not. */ + if (sgx_vtx_nmap && + sgx_vtx_slot_of(SGX_SEM_FOG, io->out_index[i]) < 0) { + slot[i] = 0; + break; + } + /* fall through */ + case SGX_SEM_GENERIC: + case SGX_SEM_TEXCOORD: + { + int want = sgx_vtx_slot_of(io->out_semantic[i], + io->out_index[i]); + + if (want >= 0) { + slot[i] = (unsigned char)want; + break; + } + } + /* By what the frame carries, not by a fixed two: + * set k is slot 2 + k, so the sets run out at + * 2 + XPSB_NSET_MAX. Written as "> 3" the third + * set's varying went to the colour slot - record + * dword 4 - while the frame declared three sets and + * the iterator read the third at dword 16, which + * nothing had written. */ + if (next_tc >= 2u + XPSB_NSET_MAX) { + /* The record holds the frame's coordinate sets + * and a colour. A varying past the sets rides + * in the colour slot, packed, and the fragment + * stage unpacks it at entry - which is the + * shape every capture has, its iterated colour + * on useissue 10 while the sets go to the + * units. Only one can: they would otherwise + * share the one colour register. */ + if (colour_used) { + fprintf(sgx_log(), "sgx: vertex output %u " + "(semantic %u) is past the " + "frame's %u coordinate sets " + "and the colour slot is " + "taken\n", i, + io->out_semantic[i], + (unsigned)XPSB_NSET_MAX); + return 0; + } + slot[i] = 1; + colour_used = 1; + break; + } + slot[i] = (unsigned char)next_tc++; + break; + default: + /* Named, so the gap is actionable: the record holds a + * position, a colour and two coordinate sets, and + * nothing else has anywhere to go. */ + fprintf(sgx_log(), "sgx: vertex output %u has semantic %u," + " which the record has no slot for\n", i, + io->out_semantic[i]); + return 0; + } + } + return 1; +} + +enum sgx_shader_status +sgx_shader_from_tgsi(struct sgx_shader *sh, int stage, + const struct tgsi_insn *insns, unsigned n, + const struct uir_binding *bindings, unsigned nbindings, + char *err, unsigned errn) +{ + return sgx_shader_from_tgsi_io(sh, stage, insns, n, bindings, + nbindings, NULL, err, errn); +} + +/* Does this fragment program do nothing but move an input to the output? + * Such a program's colour arrives from the iterator already packed, so the + * whole program is a move; anything that computes needs its four floats + * packed instead. Written out rather than assumed, because getting it the + * wrong way round is a rendering fault that looks like a hardware one. */ +static int frag_is_passthrough(const struct tgsi_insn *insns, unsigned n) +{ + unsigned i; + int moved_in = 0; + + for (i = 0; i < n; i++) { + const struct tgsi_insn *in = &insns[i]; + + if (in->dst.file != TGSI_F_OUTPUT) + continue; + /* TGSI_OPCODE_MOV, in Mesa's numbering. */ + if (in->opcode == 1 && in->nsrc == 1 && + in->src[0].file == TGSI_F_INPUT) { + moved_in = 1; + continue; + } + /* A sample straight into the output is the same case: the + * texel arrives packed and goes out as it stands. TEX, TXP + * and TXL. */ + if (in->opcode == 52 || in->opcode == 54 || in->opcode == 72) { + moved_in = 1; + continue; + } + return 0; + } + return moved_in; +} + +/* Whether the program's inputs can arrive as the iterated colour, one packed + * dword. Only a program that does nothing but move an input to the output can: + * anything that computes with an input - including deriving a texture + * coordinate from it - needs real floats, and a packed dword used as a + * coordinate samples the same texel everywhere. + * + * This is not the same question as whether the *output* is already packed. A + * program that samples into the output with a computed coordinate has a packed + * output and float inputs, and answering both with one flag made it read its + * varying out of the colour register. */ +static int frag_in_is_packed_colour(const struct tgsi_insn *insns, unsigned n) +{ + unsigned i; + int moved_in = 0; + + for (i = 0; i < n; i++) { + const struct tgsi_insn *in = &insns[i]; + + if (in->dst.file != TGSI_F_OUTPUT) + continue; + if (in->opcode != 1 || in->nsrc != 1 || + in->src[0].file != TGSI_F_INPUT) + return 0; + moved_in = 1; + } + return moved_in; +} + +/* Whether every sample in the program takes its coordinate straight from an + * iterated set, which is what the texture unit can do on its own. */ +/* The sample's coordinate may arrive through a temporary rather than straight + * from the varying: nir_to_tgsi routes it through one, and insisting on the + * input itself put every one of the X server's composites onto the + * shader-issued path - where a second texture unit does not work, so a glyph + * mask sampled as nothing and no text was drawn. A temporary written exactly + * once, by a plain move of an input, is that input under another name. */ +static int tex_coord_is_input(const struct tgsi_insn *insns, unsigned at, + const struct tgsi_src *c) +{ + unsigned i, writes = 0; + + if (c->file == TGSI_F_INPUT) + return 1; + if (c->file != TGSI_F_TEMPORARY) + return 0; + /* Every write of it, not just one: the coordinate is often moved into + * place more than once, and as long as each of them is a plain move of + * a varying the temporary still only ever holds that varying. */ + for (i = 0; i < at; i++) { + const struct tgsi_insn *w = &insns[i]; + + if (w->dst.file != TGSI_F_TEMPORARY || + w->dst.index != c->index) + continue; + writes++; + if (w->opcode != 1 /* MOV */ || w->nsrc != 1 || + w->src[0].file != TGSI_F_INPUT || + w->src[0].negate || w->src[0].absolute || + (w->src[0].swizzle & 0x0f) != (0xe4u & 0x0f)) + return 0; + } + return writes != 0; +} + +/* Which path to compile a fragment program for, when the caller has an + * opinion: -1 leaves it to the shader's own shape. The vendor's programs + * hardly ever sample for themselves - two of seventy-nine captured code + * buffers - and two units on that path do not work here, so the caller tries + * the iterated one first and only falls back when it will not build. */ +static int sgx_preiter_pref = -1; + +void sgx_shader_prefer_preiterated(int on) +{ + sgx_preiter_pref = on; +} + +/* Which two components of an input a sample takes its coordinate from. The + * fold in sgx_pipe.c put them in the sample's own swizzle. */ +static int frag_coord_comps(const struct tgsi_insn *insns, unsigned n, + unsigned in, unsigned char *c0, unsigned char *c1) +{ + unsigned i; + + for (i = 0; i < n; i++) { + const struct tgsi_src *sr = &insns[i].src[0]; + + /* A projected sample too. Matching only TEX left every one of + * Mesa's fixed-function programs - which attach a projector, + * so they are all TXP - without an answer, and the caller kept + * the default .xy. A coordinate Mesa packed at .zw was then + * fetched from the wrong two floats of the set: ioquake3's + * menu banner came out smeared and its menu text not at all. */ + if ((insns[i].opcode != 52 && insns[i].opcode != 54 && + insns[i].opcode != 72) || !insns[i].nsrc || + sr->file != TGSI_F_INPUT || sr->index != in) + continue; + *c0 = (unsigned char)(sr->swizzle & 3u); + *c1 = (unsigned char)((sr->swizzle >> 2) & 3u); + return *c0 != *c1; + } + return 0; +} + +/* The distinct coordinates the samples of one input use, as component pairs. + * Mesa packs two vec2 varyings into one vec4, so a program that samples a + * source and a mask - the shape every glyph composite has - reads .xy for one + * and .zw for the other out of a single varying. Each needs a coordinate set + * of its own, because a set is what a texture unit samples with. */ +static unsigned frag_coord_pairs(const struct tgsi_insn *insns, unsigned n, + unsigned in, unsigned char c0[2], + unsigned char c1[2]) +{ + unsigned i, np = 0; + + for (i = 0; i < n; i++) { + const struct tgsi_src *sr = &insns[i].src[0]; + unsigned char a, b, k; + + if ((insns[i].opcode != 52 && insns[i].opcode != 54) || + !insns[i].nsrc || sr->file != TGSI_F_INPUT || + sr->index != in) + continue; + a = (unsigned char)(sr->swizzle & 3u); + b = (unsigned char)((sr->swizzle >> 2) & 3u); + if (a == b) + continue; + for (k = 0; k < np; k++) + if (c0[k] == a && c1[k] == b) + break; + if (k < np) + continue; /* already have this coordinate */ + if (np == 2) + return 3; /* more than the frame can carry */ + c0[np] = a; + c1[np] = b; + np++; + } + return np; +} + +/* Whether a sample of this input reads three coordinate components - a + * volume or a cube. The set then carries the third float as a coordinate + * (UVS) rather than as a divisor (UVT), and the sample is CDIM 3. */ +/* Whether any sample in the program reads three coordinates - by its + * target (3D, cube), not by its operand's width. */ +static int frag_has_three_coord(const struct tgsi_insn *insns, unsigned n) +{ + unsigned i; + + for (i = 0; i < n; i++) + if ((insns[i].opcode == 52 || insns[i].opcode == 54 || + insns[i].opcode == 68 || insns[i].opcode == 72) && + tgsi_tex_coord_dim(insns[i].tex_target) == 3) + return 1; + return 0; +} + +/* Whether any varying is sampled with two different coordinate pairs. Mesa + * packs two vec2 texcoords into one vec4, so a masked composite reads its + * source from .xy and its mask from .zw of one input, and the iterator has + * one coordinate per set. */ +static int frag_input_volume(const struct tgsi_insn *insns, unsigned n, + unsigned in) +{ + unsigned i; + + for (i = 0; i < n; i++) + if ((insns[i].opcode == 52 || insns[i].opcode == 54 || + insns[i].opcode == 72) && insns[i].nsrc && + insns[i].src[0].file == TGSI_F_INPUT && + insns[i].src[0].index == in && + tgsi_tex_coord_dim(insns[i].tex_target) == 3) + return 1; + return 0; +} + +/* Whether a sample of this input is projected. The divide cannot be done in + * the shader: SMP reads its coordinate from the iterator and never from the + * register file, so a coordinate computed into a temporary leaves the sampler + * at texel (0,0) and every pixel comes back the same colour. The iterator + * divides by the set's third float on the way in, which the set must declare. */ +static int frag_input_projected(const struct tgsi_insn *insns, unsigned n, + unsigned in) +{ + unsigned i; + + for (i = 0; i < n; i++) + if (insns[i].opcode == 54 && insns[i].nsrc && + insns[i].src[0].file == TGSI_F_INPUT && + insns[i].src[0].index == in) + return 1; + return 0; +} + +/* Which components of an input the program reads for something other than a + * sample's coordinate. */ +static unsigned frag_input_used(const struct tgsi_insn *insns, unsigned n, + unsigned in) +{ + unsigned i, k, j, m = 0; + + for (i = 0; i < n; i++) + for (k = 0; k < insns[i].nsrc; k++) { + const struct tgsi_src *sr = &insns[i].src[k]; + + /* A sample's coordinate operand names the set rather + * than reading it. */ + if (k == 0 && (insns[i].opcode == 52 || + insns[i].opcode == 54 || + insns[i].opcode == 72)) + continue; + if (sr->file != TGSI_F_INPUT || sr->index != in) + continue; + for (j = 0; j < 4; j++) + m |= 1u << ((sr->swizzle >> (2u * j)) & 3u); + } + return m; +} + +/* Components of an input the program reads that the coordinate does not + * already carry. Reading the coordinate's own two back - which every glamor + * composite does, to offset them - still lets the set be a texture unit's; + * reading past them cannot, because the unit's texel sits where they would + * arrive. */ +static unsigned frag_in_outside(const struct tgsi_insn *insns, unsigned n, + unsigned in, int sampled) +{ + unsigned used = frag_input_used(insns, n, in); + unsigned char c0 = 0, c1 = 1; + + if (!sampled) + return used; + frag_coord_comps(insns, n, in, &c0, &c1); + return used & ~((1u << c0) | (1u << c1)); +} + +static int frag_tex_is_iterated(const struct tgsi_insn *insns, unsigned n) +{ + unsigned i; + + for (i = 0; i < n; i++) { + const struct tgsi_src *c = &insns[i].src[0]; + + /* TXP and TXL carry a projection or an explicit level, which + * the iterator does not apply. */ + if (insns[i].opcode == 54 || insns[i].opcode == 72) + return 0; + if (insns[i].opcode != 52) + continue; + if (!insns[i].nsrc || c->negate || c->absolute || + !tex_coord_is_input(insns, i, c)) + return 0; + /* Only the two channels a 2D sample reads have to be in + * place; TGSI pads the rest. */ + if ((c->swizzle & 0x0f) != (0xe4u & 0x0f)) + return 0; + /* A volume or a cube samples for itself. The iterator hands + * the unit two coordinates and a projection on this core - + * TEXPROJ_S, the form that iterates a third for the TAG, is + * SGX_FEATURE_CEM_S_USES_PROJ and reserved here (sgxdefs.h: + * 3695-3698) - and measured with the set encoded UVS the + * slice never reached the unit: TEXPROJ_T divided it by + * itself and TEXPROJ_RHW dropped it (work/feat-texunit/hw, + * hw2). The USE issue delivers all three floats, and smp3d + * reads them. */ + if (tgsi_tex_coord_dim(insns[i].tex_target) == 3) + return 0; + } + return 1; +} + +/* Whether anything the program does with a sampled texel is more than handing + * it straight to the output. Such a program needs the texel as four floats, + * because a packed dword cannot be added to or multiplied by anything. */ +static int frag_computes_with_tex(const struct tgsi_insn *insns, unsigned n) +{ + unsigned i; + + for (i = 0; i < n; i++) + if (insns[i].opcode == 52 || insns[i].opcode == 54 || + insns[i].opcode == 72) + if (insns[i].dst.file != TGSI_F_OUTPUT) + return 1; + return 0; +} + +/* Which shader inputs reach an operand: the input itself, or everything that + * flowed into the temporary it names. */ +static unsigned tgsi_src_inputs(const struct tgsi_src *src, + const unsigned *temp_in, unsigned ntemp) +{ + if (src->file == TGSI_F_INPUT) + return 1u << src->index; + if (src->file == TGSI_F_TEMPORARY && temp_in && src->index < ntemp) + return temp_in[src->index]; + return 0; +} + +/* Which sampler units carry an sRGB view, as a bitmask. GL says an sRGB + * texture is converted to linear when it is sampled, and this part has no + * sRGB texture format - the descriptor names the plain 8888 code - so the + * conversion belongs in the program, right after the sample. The mask is + * state, not part of the shader, so it is set here in the same way the + * pre-iterated preference is rather than threaded through every caller. */ +/* The sRGB mask and the iterator preference are set, used by the compile that + * follows, and cleared - which is only sound while one compile runs at a time. + * Mesa compiles shaders from more than one thread, so the whole set-compile- + * clear sequence is held under this. It is the driver's only compile-wide + * state; taking one lock around the sequence is what keeps it honest without + * threading a parameter through every entry point. */ +static simple_mtx_t sgx_compile_mtx = SIMPLE_MTX_INITIALIZER; + +void sgx_shader_compile_lock(void) +{ + simple_mtx_lock(&sgx_compile_mtx); +} + +void sgx_shader_compile_unlock(void) +{ + simple_mtx_unlock(&sgx_compile_mtx); +} + +static unsigned sgx_srgb_mask; + +void sgx_shader_srgb_units(unsigned mask) +{ + sgx_srgb_mask = mask; +} + +unsigned sgx_shader_srgb_units_get(void) +{ + return sgx_srgb_mask; +} + +static unsigned sgx_depth_mask; + +void sgx_shader_depth_units(unsigned mask) +{ + sgx_depth_mask = mask; +} + +unsigned sgx_shader_depth_units_get(void) +{ + return sgx_depth_mask; +} + +static int sgx_face_swap; + +void sgx_shader_face_swap(int swap) +{ + sgx_face_swap = swap != 0; +} + +/* gl_FrontFacing without a varying: the ISP sets bit 0 of the USE's global + * register BFCONTROL for a triangle wound clockwise in device space, and the + * vendor's compiler reads it with a bitwise test (usc2/icvt_core.c + * CheckFaceType; icvt_f32.c UF_MISC_FACETYPE gives +1.0/-1.0), then negates + * the answer under a "swap front face" uniform the GLES driver sets from + * glFrontFace and the surface's Y inversion (opengles2/uniform.c + * GLSLBV_PMXSWAPFRONTFACE). Here that uniform is a compile-time sign, since + * the program is rebuilt when it changes: UIR_FACE is +1.0 where the bit is + * clear, and sgx_face_swap_of() says whether that is the front. Measured on + * the booklet (b45eef0): under GL_CCW with no flip the bit is set for the + * front-facing quad, the opposite of what the cull word's derivation + * predicted, so the sign is front_ccw XOR the target's flip rather than its + * inverse. + * + * The input is read from a fresh register the face is computed into ahead + * of everything, and it takes no coordinate set, so it works on both vertex + * paths: the ISP produces the bit whichever stage transformed the vertex. */ +static int sgx_face_apply(struct uir_shader *s, unsigned in, int swap) +{ + uint32_t v = uir_alloc_virt(s); + size_t k, old = s->ninsns, add = swap ? 2u : 1u; + struct uir_insn *o; + + if (!uir_emit(s, UIR_FACE)) + return -1; + if (swap && !uir_emit(s, UIR_MUL)) + return -1; + /* Ahead of the program: the value is read anywhere in it. */ + memmove(s->insns + add, s->insns, old * sizeof *s->insns); + o = &s->insns[0]; + memset(o, 0, sizeof *o); + o->op = UIR_FACE; + o->type = UIR_F32; + o->writemask = 0xf; + o->dst = uir_reg(UIR_REG_VIRT, v); + if (swap) { + o = &s->insns[1]; + memset(o, 0, sizeof *o); + o->op = UIR_MUL; + o->type = UIR_F32; + o->writemask = 0xf; + o->dst = uir_reg(UIR_REG_VIRT, v); + o->src[0] = uir_reg(UIR_REG_VIRT, v); + o->src[1] = uir_imm1(-1.0f); + o->nsrc = 2; + } + for (k = add; k < s->ninsns; k++) { + unsigned j; + + for (j = 0; j < s->insns[k].nsrc; j++) { + struct uir_ref *r = &s->insns[k].src[j]; + + if (r->cls == UIR_REG_IN && r->index == in) { + r->cls = UIR_REG_VIRT; + r->index = v; + } + } + } + return 0; +} + +/* The texture unit's REPEAT wraps at a power of two rather than at the + * texture's own width: a 40-wide texture repeats every 64 texels and the 24 + * between are padding. The coordinate is scaled by the real width, though - + * measured on the booklet with driver/test/sgx_egl_npot, where every + * power-of-two width is exact and 24 wraps at 32, 40 at 64, 96 at 128 - so + * the program's own fractional part is the whole of REPEAT, with no scale to + * undo afterwards. + * + * The vendor never meets this: texmgmt.c forces CLAMP on a non-power-of-two + * texture, so a repeating sample of one cannot arise on its stack. This + * driver advertises ARB_texture_non_power_of_two, so the wrap is done in the + * program instead of taken away from the caller. + * + * A projective sample is left alone. Its coordinate is divided by w after + * this point, so a fraction taken here would be the fraction of the wrong + * number; that case keeps the hardware wrap and is still wrong for a + * non-power-of-two texture. + * + * Linear filtering has one texel of seam: at the top of the range the unit + * blends the last texel with the padding beside it rather than with texel + * zero, which REPEAT would have wrapped to. Point sampling is exact. */ +static unsigned sgx_npot_mask; + +void sgx_shader_npot_units(unsigned mask) +{ + sgx_npot_mask = mask; +} + +unsigned sgx_shader_npot_units_get(void) +{ + return sgx_npot_mask; +} + +/* The coordinate components of one sample that the program has to wrap, as a + * writemask: bit 0 is s, bit 1 is t. */ +static unsigned npot_site(const struct uir_insn *n, unsigned mask, + unsigned *mirror) +{ + unsigned j; + + *mirror = 0; + if (n->op != UIR_TEX && n->op != UIR_TEXLOD) + return 0; + for (j = 0; j < n->nsrc; j++) { + unsigned i = n->src[j].index; + + if (n->src[j].cls != UIR_REG_SAMPLER) + continue; + if (i >= 8u) + return 0; + *mirror = ((SGX_NPOT_MIRROR_S(mask) >> i) & 1u) + | (((SGX_NPOT_MIRROR_T(mask) >> i) & 1u) << 1); + return ((SGX_NPOT_REPEAT_S(mask) >> i) & 1u) + | (((SGX_NPOT_REPEAT_T(mask) >> i) & 1u) << 1); + } + return 0; +} + +static unsigned npot_cost(unsigned repeat, unsigned mirror) +{ + if (!repeat && !mirror) + return 0; + return 1u + (repeat ? 1u : 0u) + (mirror ? 5u : 0u); +} + +static void npot_insn(struct uir_insn *o, enum uir_op op, uint32_t dst, + unsigned mask, unsigned nsrc) +{ + memset(o, 0, sizeof *o); + o->op = op; + o->type = UIR_F32; + o->dst = uir_reg(UIR_REG_VIRT, dst); + o->writemask = (uint8_t)mask; + o->nsrc = nsrc; +} + +/* Returns how many samples were given their own wrap, or -1. The count is + * what tells the caller the coordinate is computed now: a sample the iterator + * serves reads the varying the rasteriser produced, which a fraction of it is + * not, so the two cannot both be chosen. */ +static int sgx_npot_wrap(struct uir_shader *s, unsigned mask) +{ + size_t r, w, add = 0; + int sites = 0; + uint32_t t, t2; + + if (!mask || !s) + return 0; + for (r = 0; r < s->ninsns; r++) { + unsigned mir, rep = npot_site(&s->insns[r], mask, &mir); + + if (npot_cost(rep, mir)) + sites++; + add += npot_cost(rep, mir); + } + if (!add) + return 0; + + t = uir_alloc_virt(s); + t2 = uir_alloc_virt(s); + + /* Reserve the room through the shader's own growth, then fill from the + * back so nothing is overwritten before it has been moved. */ + w = s->ninsns; + for (r = 0; r < add; r++) + if (!uir_emit(s, UIR_MOV)) + return -1; + r = w; + w = s->ninsns; + while (r > 0) { + struct uir_insn n = s->insns[--r]; + unsigned mir, rep = npot_site(&n, mask, &mir); + unsigned cost = npot_cost(rep, mir), k = 0; + struct uir_ref coord = n.src[0]; + struct uir_insn *o; + + if (cost) + n.src[0] = uir_reg(UIR_REG_VIRT, t); + s->insns[--w] = n; + if (!cost) + continue; + w -= cost; + o = &s->insns[w]; + /* The coordinate as the program computed it, so that whatever + * swizzle it carried is applied once and the sample then reads + * one register. */ + npot_insn(&o[k], UIR_MOV, t, 0xf, 1); + o[k++].src[0] = coord; + if (rep) { + npot_insn(&o[k], UIR_FRC, t, rep, 1); + o[k++].src[0] = uir_reg(UIR_REG_VIRT, t); + } + if (mir) { + /* min(m, 2 - m) over m = frac(u / 2) * 2, which is the + * mirror without a branch. The second term is a MAD + * rather than a subtraction from an immediate: an + * immediate is a source the backend folds, not one it + * takes first. */ + npot_insn(&o[k], UIR_MUL, t, mir, 2); + o[k].src[0] = uir_reg(UIR_REG_VIRT, t); + o[k++].src[1] = uir_imm1(0.5f); + npot_insn(&o[k], UIR_FRC, t, mir, 1); + o[k++].src[0] = uir_reg(UIR_REG_VIRT, t); + npot_insn(&o[k], UIR_MUL, t, mir, 2); + o[k].src[0] = uir_reg(UIR_REG_VIRT, t); + o[k++].src[1] = uir_imm1(2.0f); + npot_insn(&o[k], UIR_MAD, t2, mir, 3); + o[k].src[0] = uir_reg(UIR_REG_VIRT, t); + o[k].src[1] = uir_imm1(-1.0f); + o[k++].src[2] = uir_imm1(2.0f); + npot_insn(&o[k], UIR_MIN, t, mir, 2); + o[k].src[0] = uir_reg(UIR_REG_VIRT, t); + o[k++].src[1] = uir_reg(UIR_REG_VIRT, t2); + } + } + return sites; +} + +/* Whether a vertex program compiled now should pack its colour output into + * one dword. It belongs to the pair rather than to the program - the same + * vertex shader is right either way and only a fragment program that takes a + * packed colour needs it - so the driver compiles a second form with this set, + * the same way the sRGB mask is handed over rather than threaded through. */ +static int sgx_vtx_pack_colour; + +void sgx_shader_vtx_pack_colour(int on) +{ + sgx_vtx_pack_colour = on; +} + +/* Where the frame puts each varying, by the semantic that carries it. + * + * The frame numbers its coordinate sets sampled first, because the set index + * is the texissue and the part samples only the low sets. Numbering the + * transform's outputs by semantic in declaration order instead agrees with + * that only while the sampled varyings happen to be declared first, and where + * it does not the transform writes each varying into the other's place. The + * driver knows both programs when it binds them, so it says here which slot + * each semantic belongs in and the transform follows the frame rather than + * guessing. Empty means no pairing has been settled and the old numbering + * stands. */ +static unsigned char sgx_vtx_sem[16], sgx_vtx_idx[16], sgx_vtx_slot[16]; +static unsigned char sgx_vtx_dw[16]; +static int sgx_vtx_have_dw; + +void sgx_shader_vtx_layout(const unsigned char *sem, const unsigned char *idx, + const unsigned char *slot, unsigned n) +{ + sgx_shader_vtx_layout_dw(sem, idx, slot, NULL, n); +} + +/* The same, with the emitted dword each output starts at. A slot is a quad; + * the iterator packs the coordinate sets by their own widths behind the eight + * fixed dwords, so the two part company as soon as a set is narrower than four + * floats, and the caller has to say where it really put them. */ +void sgx_shader_vtx_layout_dw(const unsigned char *sem, + const unsigned char *idx, + const unsigned char *slot, + const unsigned char *dw, unsigned n) +{ + sgx_vtx_nmap = 0; + sgx_vtx_have_dw = dw != NULL; + if (!sem || !idx || !slot) + return; + for (; sgx_vtx_nmap < n && sgx_vtx_nmap < 16; sgx_vtx_nmap++) { + sgx_vtx_sem[sgx_vtx_nmap] = sem[sgx_vtx_nmap]; + sgx_vtx_idx[sgx_vtx_nmap] = idx[sgx_vtx_nmap]; + sgx_vtx_slot[sgx_vtx_nmap] = slot[sgx_vtx_nmap]; + sgx_vtx_dw[sgx_vtx_nmap] = dw ? dw[sgx_vtx_nmap] : + (unsigned char)(4u * slot[sgx_vtx_nmap]); + } +} + +/* The semantic index as well as the name: two varyings of one semantic - + * generic 0 and generic 1, which is most of what Mesa emits - both matched the + * first entry and were given the same slot. */ +/* The emitted dword for a semantic, falling back to the slot's quad. */ +static unsigned char sgx_vtx_dw_of(unsigned sem, unsigned idx, + unsigned char slot) +{ + unsigned q; + + for (q = 0; q < sgx_vtx_nmap; q++) + if (sgx_vtx_sem[q] == sem && sgx_vtx_idx[q] == idx) + return sgx_vtx_dw[q]; + return (unsigned char)(4u * slot); +} + +static int sgx_vtx_slot_of(unsigned sem, unsigned idx) +{ + unsigned q; + + for (q = 0; q < sgx_vtx_nmap; q++) + if (sgx_vtx_sem[q] == sem && sgx_vtx_idx[q] == idx) + return sgx_vtx_slot[q]; + return -1; +} + +/* c <= 0.04045 ? c / 12.92 : ((c + 0.055) / 1.055) ^ 2.4, on rgb only - + * alpha is linear by definition. Seven instructions, and the power is + * exp2(2.4 * log2(x)) because that is what the part has. The argument to the + * logarithm is (c + 0.055) / 1.055, which is at least 0.052 for any texel, so + * it never reaches log2(0). */ +#define SGX_SRGB_INSNS 7u + +static int srgb_site(const struct uir_insn *n, unsigned mask) +{ + unsigned j; + + if (n->op != UIR_TEX && n->op != UIR_TEXPROJ && n->op != UIR_TEXLOD && + n->op != UIR_TEXBIAS) + return 0; + for (j = 0; j < n->nsrc; j++) + if (n->src[j].cls == UIR_REG_SAMPLER) + return n->src[j].index < 32 && + (mask >> n->src[j].index) & 1; + return 0; +} + +static void srgb_insn(struct uir_insn *o, enum uir_op op, uint32_t dst, + unsigned nsrc) +{ + memset(o, 0, sizeof *o); + o->op = op; + o->type = UIR_F32; + o->dst = uir_reg(UIR_REG_VIRT, dst); + o->writemask = 0x7; /* rgb; alpha stays as sampled */ + o->nsrc = nsrc; +} + +/* Returns 0, or -1 if the conversion could not be built. A sample that writes + * straight to an output - the shape a shader most naturally has, sample and + * done - is retargeted to a register of its own, the alpha it fetched is moved + * out untouched, and the decoded colour is written over it. */ +static int sgx_srgb_decode(struct uir_shader *s, unsigned mask) +{ + size_t r, w, nd = 0, nout = 0, iout; + uint32_t t0, t1, t2, *outv = NULL; + + if (!mask || !s) + return 0; + for (r = 0; r < s->ninsns; r++) + if (srgb_site(&s->insns[r], mask)) { + if (s->insns[r].dst.cls != UIR_REG_VIRT) + nout++; + nd++; + } + if (!nd) + return 0; + + /* One per retargeted sample, taken before the shader grows. */ + if (nout) { + outv = malloc(nout * sizeof *outv); + if (!outv) + return -1; + for (r = 0; r < nout; r++) + outv[r] = uir_alloc_virt(s); + } + t0 = uir_alloc_virt(s); + t1 = uir_alloc_virt(s); + t2 = uir_alloc_virt(s); + + /* Reserve the room through the shader's own growth, then fill it from + * the back so nothing is overwritten before it has been moved. */ + w = s->ninsns; + for (r = 0; r < SGX_SRGB_INSNS * nd + nout; r++) + if (!uir_emit(s, UIR_MOV)) { + free(outv); + return -1; + } + r = w; + w = s->ninsns; + iout = nout; + while (r > 0) { + struct uir_insn n = s->insns[--r]; + + if (srgb_site(&n, mask)) { + struct uir_ref out = n.dst; + unsigned blk = SGX_SRGB_INSNS; + struct uir_ref v; + struct uir_insn *o; + + /* Filled back to front, so the registers are consumed + * from the end of the array the forward scan filled. */ + if (out.cls != UIR_REG_VIRT) { + n.dst = uir_reg(UIR_REG_VIRT, outv[--iout]); + blk++; + } + v = uir_reg(UIR_REG_VIRT, n.dst.index); + o = &s->insns[w - blk]; + w -= blk; + srgb_insn(&o[0], UIR_MAD, t0, 3); + o[0].src[0] = v; + o[0].src[1] = uir_imm1(1.0f / 1.055f); + o[0].src[2] = uir_imm1(0.055f / 1.055f); + srgb_insn(&o[1], UIR_LOG2, t0, 1); + o[1].src[0] = uir_reg(UIR_REG_VIRT, t0); + srgb_insn(&o[2], UIR_MUL, t0, 2); + o[2].src[0] = uir_reg(UIR_REG_VIRT, t0); + o[2].src[1] = uir_imm1(2.4f); + srgb_insn(&o[3], UIR_EXP2, t0, 1); + o[3].src[0] = uir_reg(UIR_REG_VIRT, t0); + srgb_insn(&o[4], UIR_MUL, t1, 2); + o[4].src[0] = v; + o[4].src[1] = uir_imm1(1.0f / 12.92f); + srgb_insn(&o[5], UIR_ADD, t2, 2); + o[5].src[0] = v; + o[5].src[1] = uir_imm1(-0.04045f); + srgb_insn(&o[6], UIR_CMP, n.dst.index, 3); + o[6].src[0] = uir_reg(UIR_REG_VIRT, t2); + o[6].src[1] = uir_reg(UIR_REG_VIRT, t1); + o[6].src[2] = uir_reg(UIR_REG_VIRT, t0); + /* Every step above writes a register; the retargeted + * sample's own output is written once, at the end, + * with the alpha the decode left untouched. */ + if (out.cls != UIR_REG_VIRT) { + srgb_insn(&o[7], UIR_MOV, 0, 1); + o[7].dst = out; + o[7].writemask = 0xf; + o[7].src[0] = v; + } + } + s->insns[--w] = n; + } + free(outv); + return 0; +} + +/* A depth surface sampled as the four bytes it is. The texture unit has no + * depth format at any width on this core - EURASIA_PDS_DOUTT1_TEXFORMAT_* + * runs U8 to F32 with no depth row, and the X8U24/U8U24 codes at 21 and 22 + * are behind SGX545/543/544/554 (sgxdefs.h:4607-4619) - and the ISP stores + * depth as I24ZI8S, a 24-bit integer with the stencil above it (ZLSCTL + * [19:18] = 1, the 0x00450000 the raster stream carries; measured at a window + * depth of 0.25 as the 0x00400000 in xpsb_frame.c:3129-3140). So the surface + * is described as the 32-bit BGRA it is - the vendor's own row for D24X8, + * sgx535pixfmts.h:774 - and the depth is put back together here. + * + * The unit hands the texel over as four bytes in one register, unpacked to + * floats byte 2, 1, 0, 3 into x, y, z, w (usse_cg.c:2686). The dword is + * little-endian, so x is z[23:16], y is z[15:8] and z is z[7:0]: + * + * d = (x * 65536 + y * 256 + z) * 255 / (2^24 - 1) + * = x * 65536/65793 + y * 256/65793 + z * 1/65793 + * + * whose three coefficients sum to one exactly, so a fully set depth reads + * 1.0 rather than one ulp below it. The result goes to all four channels + * as GL says a depth sample reads - (d, d, d, 1) - so it is right whatever + * the view swizzle then makes of it, including the identity. */ +#define SGX_DEPTH_INSNS 5u +#define SGX_DEPTH_SCALE 65793.0f + +static int depth_site(const struct uir_insn *n, unsigned mask) +{ + unsigned j; + + if (n->op != UIR_TEX && n->op != UIR_TEXPROJ && n->op != UIR_TEXLOD && + n->op != UIR_TEXBIAS) + return 0; + for (j = 0; j < n->nsrc; j++) + if (n->src[j].cls == UIR_REG_SAMPLER) + return n->src[j].index < 32 && + (mask >> n->src[j].index) & 1; + return 0; +} + +static void depth_insn(struct uir_insn *o, enum uir_op op, uint32_t dst, + unsigned mask, unsigned nsrc) +{ + memset(o, 0, sizeof *o); + o->op = op; + o->type = UIR_F32; + o->dst = uir_reg(UIR_REG_VIRT, dst); + o->writemask = (uint8_t)mask; + o->nsrc = nsrc; +} + +/* Returns 0, or -1 if the expansion could not be built. A sample that writes + * straight to an output is retargeted to a register of its own and moved out + * once the depth has been assembled, exactly as the sRGB decode does. */ +static int sgx_depth_expand(struct uir_shader *s, unsigned mask) +{ + size_t r, w, nd = 0, nout = 0, iout; + uint32_t t0, *outv = NULL; + + if (!mask || !s) + return 0; + for (r = 0; r < s->ninsns; r++) + if (depth_site(&s->insns[r], mask)) { + if (s->insns[r].dst.cls != UIR_REG_VIRT) + nout++; + nd++; + } + if (!nd) + return 0; + + if (nout) { + outv = malloc(nout * sizeof *outv); + if (!outv) + return -1; + for (r = 0; r < nout; r++) + outv[r] = uir_alloc_virt(s); + } + t0 = uir_alloc_virt(s); + + w = s->ninsns; + for (r = 0; r < SGX_DEPTH_INSNS * nd + nout; r++) + if (!uir_emit(s, UIR_MOV)) { + free(outv); + return -1; + } + r = w; + w = s->ninsns; + iout = nout; + while (r > 0) { + struct uir_insn n = s->insns[--r]; + + if (depth_site(&n, mask)) { + struct uir_ref out = n.dst; + unsigned blk = SGX_DEPTH_INSNS; + struct uir_ref v; + struct uir_insn *o; + + if (out.cls != UIR_REG_VIRT) { + n.dst = uir_reg(UIR_REG_VIRT, outv[--iout]); + blk++; + } + v = uir_reg(UIR_REG_VIRT, n.dst.index); + o = &s->insns[w - blk]; + w -= blk; + /* Low byte first, so the two multiply-adds above it + * accumulate into one register. */ + depth_insn(&o[0], UIR_MUL, t0, 0x1, 2); + o[0].src[0] = uir_swz(v, UIR_SWZ(2, 2, 2, 2)); + o[0].src[1] = uir_imm1(1.0f / SGX_DEPTH_SCALE); + depth_insn(&o[1], UIR_MAD, t0, 0x1, 3); + o[1].src[0] = uir_swz(v, UIR_SWZ(1, 1, 1, 1)); + o[1].src[1] = uir_imm1(256.0f / SGX_DEPTH_SCALE); + o[1].src[2] = uir_reg(UIR_REG_VIRT, t0); + depth_insn(&o[2], UIR_MAD, t0, 0x1, 3); + o[2].src[0] = uir_swz(v, UIR_SWZ_XXXX); + o[2].src[1] = uir_imm1(65536.0f / SGX_DEPTH_SCALE); + o[2].src[2] = uir_reg(UIR_REG_VIRT, t0); + /* The whole depth is assembled in a register of its + * own and moved back over the texel, rather than + * permuting the texel where it lies: reading and + * writing one register under a swizzle is the + * write-after-read the backend has to break up. */ + depth_insn(&o[3], UIR_MOV, n.dst.index, 0x7, 1); + o[3].src[0] = uir_swz(uir_reg(UIR_REG_VIRT, t0), + UIR_SWZ_XXXX); + depth_insn(&o[4], UIR_MOV, n.dst.index, 0x8, 1); + o[4].src[0] = uir_imm1(1.0f); + if (out.cls != UIR_REG_VIRT) { + depth_insn(&o[5], UIR_MOV, 0, 0xf, 1); + o[5].dst = out; + o[5].src[0] = v; + } + } + s->insns[--w] = n; + } + free(outv); + return 0; +} + +/* The view's swizzle, applied to what the part fetched. Held while a program + * is compiled, the same way the sRGB mask is. */ +static uint16_t *sgx_swz_units_arr; +static unsigned sgx_swz_units_n; + +void sgx_shader_swz_units(const uint16_t *swz, unsigned n) +{ + unsigned i; + + if (n > sgx_swz_units_n) { + uint16_t *p = realloc(sgx_swz_units_arr, n * sizeof *p); + + if (!p) + return; + sgx_swz_units_arr = p; + } + sgx_swz_units_n = n; + for (i = 0; i < n; i++) + sgx_swz_units_arr[i] = swz ? swz[i] : SGX_SWZ_IDENTITY; +} + +static int swz_any_nonidentity(void) +{ + unsigned i; + + for (i = 0; i < sgx_swz_units_n; i++) + if (sgx_swz_units_arr[i] != SGX_SWZ_IDENTITY) + return 1; + return 0; +} + +static unsigned swz_of_unit(const uint16_t *k, unsigned n, unsigned u) +{ + return u < n ? k[u] : (unsigned)SGX_SWZ_IDENTITY; +} + +/* The texel class of each unit's view, held for the compile the same way. */ +static unsigned char *sgx_cls_units_arr; +static unsigned sgx_cls_units_n; + +void sgx_shader_cls_units(const unsigned char *cls, unsigned n) +{ + unsigned i; + + if (n > sgx_cls_units_n) { + unsigned char *p = realloc(sgx_cls_units_arr, n); + + if (!p) + return; + sgx_cls_units_arr = p; + } + sgx_cls_units_n = n; + for (i = 0; i < n; i++) + sgx_cls_units_arr[i] = cls ? cls[i] : 0; +} + +static unsigned cls_of_unit(const unsigned char *k, unsigned n, unsigned u) +{ + return u < n ? k[u] : 0u; +} + +int sgx_shader_cls_differs(const struct sgx_shader *sh, + const unsigned char *cls, unsigned n) +{ + unsigned i; + + if (!sh || !sh->sampler_mask) + return 0; + for (i = 0; i < n || i < sh->ncls_key; i++) { + if (i >= 32 || !((sh->sampler_mask >> i) & 1u)) + continue; + if (cls_of_unit(cls, n, i) != + cls_of_unit(sh->cls_key, sh->ncls_key, i)) + return 1; + } + return 0; +} + +/* Which state block a unit's sample reads first, and how many planes the + * program was built to sample there. A unit past the ones the program names + * - a bound view the program never reads - gets a block of its own behind + * the last, so writing its descriptor lands nowhere the code reads from. */ +static unsigned smp_unit_slot(const struct sgx_shader *sh, unsigned unit, + unsigned *nchunks) +{ + unsigned n = 1, slot; + + if (sh && sh->smp_slot && unit < sh->nsamp) { + slot = sh->smp_slot[unit]; + n = (unit + 1u < sh->nsamp ? sh->smp_slot[unit + 1] : + sh->nsmp_slots) - slot; + if (!n) + n = 1; + } else { + slot = (sh ? sh->nsmp_slots : 0u) + + (unit - (sh ? sh->nsamp : 0u)); + } + if (nchunks) + *nchunks = n; + return slot; +} + +unsigned sgx_shader_smp_reg(const struct sgx_shader *sh, unsigned unit, + unsigned chunk, unsigned nu, unsigned *nchunks) +{ + unsigned base; + + if (nchunks) + *nchunks = 1; + if (!sh) + return 0; + /* What the backend reported, which is where its SMP instructions + * actually read. Zero is a base - a program with no uniforms puts + * its first block at sa0 - so what says the backend reported a + * layout is that it laid slots down at all, not that the base is + * non-zero. Deriving it here instead got ioquake3's menu wrong both + * ways, and the float textures four registers wrong. */ + if (sh->nsmp_slots && !getenv("SGX_NO_SMP_BASE")) + base = sh->smp_base; + else if (sh->pool_base >= nu * 4u) + base = sh->pool_base - nu * 4u; + else + base = sh->nuniform * 4u; + return base + SGX_SMP_SLOT_REGS * + (smp_unit_slot(sh, unit, nchunks) + chunk); +} + +int sgx_shader_cmp_differs(const struct sgx_shader *sh, const uint32_t *cmp, + unsigned n) +{ + unsigned i; + + if (!sh || !sh->src_nir) + return 0; + for (i = 0; i < n || i < sh->ncmp_key; i++) { + uint32_t a = i < n ? cmp[i] : 0u; + uint32_t b = i < sh->ncmp_key ? sh->cmp_key[i] : 0u; + + if (i >= 32 || !((sh->shadow_mask >> i) & 1u)) + continue; + if (a != b) + return 1; + } + return 0; +} + +/* The sampler a texture instruction reads, and whether its view asks for + * anything other than the four channels in order. */ +static int swz_site(const struct uir_insn *n, const uint16_t *k, unsigned nk, + unsigned *key) +{ + unsigned j; + + if (n->op != UIR_TEX && n->op != UIR_TEXPROJ && n->op != UIR_TEXLOD && + n->op != UIR_TEXBIAS) + return 0; + for (j = 0; j < n->nsrc; j++) + if (n->src[j].cls == UIR_REG_SAMPLER) { + *key = swz_of_unit(k, nk, n->src[j].index); + return *key != SGX_SWZ_IDENTITY; + } + return 0; +} + +/* Which destination channels come from the sample, which are constant, and + * the swizzle that gathers the sampled ones. */ +static void swz_masks(unsigned key, unsigned *tex, unsigned *zero, + unsigned *one, uint8_t *perm) +{ + unsigned i; + + *tex = *zero = *one = 0; + *perm = 0; + for (i = 0; i < 4; i++) { + unsigned c = SGX_SWZ_CHAN(key, i); + + if (c == SGX_SWZ_ZERO) { + *zero |= 1u << i; + } else if (c == SGX_SWZ_ONE) { + *one |= 1u << i; + } else { + *tex |= 1u << i; + *perm |= (uint8_t)(c << (i * 2)); + } + } +} + +static void swz_insn(struct uir_insn *o, struct uir_ref dst, unsigned mask) +{ + memset(o, 0, sizeof *o); + o->op = UIR_MOV; + o->type = UIR_F32; + o->dst = dst; + o->writemask = (uint8_t)mask; + o->nsrc = 1; +} + +/* Returns 0, or -1 if a swizzled sample writes somewhere this cannot rewrite. + * + * The gather goes through a fresh register rather than permuting the sample + * in place: reading and writing one register under a swizzle is the + * write-after-read the backend has to break up, and there is no reason to + * hand it that when a move costs the same. */ +static int sgx_swz_apply(struct uir_shader *s, const uint16_t *k, unsigned nk) +{ + size_t r, w, add = 0; + unsigned key; + + /* Rebuild but insert nothing, to tell the two apart when a swizzle + * makes something render wrongly. */ + if (!s || !k || !nk || getenv("SGX_SWZ_NOCODE")) + return 0; + for (r = 0; r < s->ninsns; r++) + if (swz_site(&s->insns[r], k, nk, &key)) { + unsigned tex, zero, one; + uint8_t perm; + + swz_masks(key, &tex, &zero, &one, &perm); + add += (tex ? 1u : 0u) + (zero ? 1u : 0u) + + (one ? 1u : 0u); + } + if (!add) + return 0; + + /* Reserve the room through the shader's own growth, then fill from the + * back so nothing is overwritten before it has been moved. */ + w = s->ninsns; + for (r = 0; r < add; r++) + if (!uir_emit(s, UIR_MOV)) + return -1; + r = w; + w = s->ninsns; + while (r > 0) { + struct uir_insn n = s->insns[--r]; + + if (swz_site(&n, k, nk, &key)) { + unsigned tex, zero, one, cnt; + struct uir_ref dst = n.dst, t; + struct uir_insn *o; + uint8_t perm; + + swz_masks(key, &tex, &zero, &one, &perm); + cnt = (tex ? 1u : 0u) + (zero ? 1u : 0u) + + (one ? 1u : 0u); + /* The sample is redirected into a register of its own + * and the moves write where it was going. That is what + * lets this rewrite a sample straight into the colour + * output - "gl_FragColor = texture2D(...)" - which is + * the commonest shape of all and cannot be permuted in + * place. */ + t = uir_reg(UIR_REG_VIRT, uir_alloc_virt(s)); + n.dst = t; + n.writemask = 0xf; + w -= cnt; + o = &s->insns[w]; + if (tex) { + swz_insn(&o[0], dst, tex); + o[0].src[0] = uir_swz(t, perm); + o++; + } + if (zero) { + swz_insn(&o[0], dst, zero); + o[0].src[0] = uir_imm1(0.0f); + o++; + } + if (one) { + swz_insn(&o[0], dst, one); + o[0].src[0] = uir_imm1(1.0f); + } + } + s->insns[--w] = n; + } + return 0; +} + +int sgx_shader_swz_differs(const struct sgx_shader *sh, const uint16_t *swz, + unsigned n) +{ + unsigned i; + + if (!sh || !sh->sampler_mask) + return 0; + for (i = 0; i < n || i < sh->nswz_key; i++) { + if (i >= 32 || !((sh->sampler_mask >> i) & 1u)) + continue; + /* A shadow sampler's swizzle went into its comparison's + * lowering, ahead of this; here it reads as the identity. */ + if (i < sh->ncmp_key && sh->cmp_key[i]) + continue; + if (swz_of_unit(swz, n, i) != + swz_of_unit(sh->swz_key, sh->nswz_key, i)) + return 1; + } + return 0; +} + +/* The record's colour slot goes out blue first. + * + * The frame's iterator packs the colour slot's four floats into one 8888 + * dword, first float into the low byte, and an A8R8G8B8 target reads that + * byte as blue - so the draw module writes its record z,y,x,w + * (sgx_pipe_vbuf.c, measured). A hardware vertex program writes the slot in + * the order the shader computed it, which put red in the low byte instead + * and swapped red and blue on every draw the part transformed. The fragment + * stage reads one convention for both paths, so the swap belongs here. + * + * The colour output is renamed to a virtual register and a closing move + * permutes it, rather than the writes being permuted in place: only a + * component-wise operation can carry the permutation in its own operands, + * and what writes the slot is not restricted to those. */ +static int sgx_vs_colour_bgra(struct uir_shader *s, uint32_t out_index) +{ + size_t i, at, orig; + unsigned mask = 0; + uint32_t t; + struct uir_insn *n; + + for (i = 0; i < s->ninsns; i++) + if (s->insns[i].dst.cls == UIR_REG_OUT && + s->insns[i].dst.index == out_index) + mask |= s->insns[i].writemask; + if (!mask) + return 0; + + t = uir_alloc_virt(s); + for (i = 0; i < s->ninsns; i++) { + struct uir_insn *o = &s->insns[i]; + unsigned k; + + if (o->dst.cls == UIR_REG_OUT && o->dst.index == out_index) { + o->dst.cls = UIR_REG_VIRT; + o->dst.index = t; + } + for (k = 0; k < o->nsrc; k++) + if (o->src[k].cls == UIR_REG_OUT && + o->src[k].index == out_index) { + o->src[k].cls = UIR_REG_VIRT; + o->src[k].index = t; + } + } + + orig = s->ninsns; + at = orig; + while (at > 0 && s->insns[at - 1].op == UIR_EMIT) + at--; + if (!uir_emit(s, UIR_MOV)) + return -1; + memmove(&s->insns[at + 1], &s->insns[at], + (orig - at) * sizeof s->insns[0]); + n = &s->insns[at]; + memset(n, 0, sizeof *n); + n->op = UIR_MOV; + n->type = UIR_F32; + n->dst = uir_reg(UIR_REG_OUT, out_index); + /* x and z trade places, so the channels written trade with them. */ + n->writemask = (uint8_t)((mask & 0xau) | ((mask & 1u) << 2) | + ((mask >> 2) & 1u)); + n->nsrc = 1; + n->src[0] = uir_swz(uir_reg(UIR_REG_VIRT, t), UIR_SWZ(2, 1, 0, 3)); + return 0; +} + +enum sgx_shader_status +sgx_shader_from_tgsi_io(struct sgx_shader *sh, int stage, + const struct tgsi_insn *insns, unsigned n, + const struct uir_binding *bindings, unsigned nbindings, + const struct tgsi_io *io, char *err, unsigned errn) +{ + struct uir_shader *s; + uint32_t tex_smp[16]; + /* As tex_smp, but only for a sampler whose coordinate is the input + * itself rather than something computed from it. */ + uint32_t tex_dsmp[16]; + unsigned char tex_direct[16]; + unsigned io_nin; + int smp_and_read; + struct uir_codegen_opts opts; + struct uir_codegen_result res; + char cgerr[256] = ""; + unsigned at = 0, i, d = 0, has_tex = 0, ntemp = 0; + unsigned *temp_in; + unsigned char *tex_class = NULL, *tex_chunks = NULL; + int wide_texel = 0; + int npot_sites = 0; + enum tgsi_xlat_status xs; + int rc = -1; + + if (!sh || !insns || !n) + return SGX_SHADER_TRANSLATE_FAILED; + /* Recompiling into the same object must not leak the last buffer, and + * must not drop the source it would be rebuilt from again. */ + { + struct tgsi_insn *keep_insns = sh->src_insns; + struct tgsi_io *keep_io = sh->src_io; + unsigned keep_n = sh->src_n; + uint16_t *keep_swz = sh->swz_key; + unsigned keep_nswz = sh->nswz_key; + unsigned char *keep_cls = sh->cls_key; + unsigned keep_ncls = sh->ncls_key; + uint32_t *keep_cmp = sh->cmp_key; + unsigned keep_ncmp = sh->ncmp_key; + void *keep_nir = sh->src_nir; + unsigned keep_shadow = sh->shadow_mask; + + sgx_shader_fini(sh); + memset(sh, 0, sizeof(*sh)); + sh->src_insns = keep_insns; + sh->src_io = keep_io; + sh->src_n = keep_n; + sh->swz_key = keep_swz; + sh->nswz_key = keep_nswz; + sh->cls_key = keep_cls; + sh->ncls_key = keep_ncls; + sh->cmp_key = keep_cmp; + sh->ncmp_key = keep_ncmp; + sh->src_nir = keep_nir; + sh->shadow_mask = keep_shadow; + } + sh->srgb_key = sgx_srgb_mask; + sh->depth_key = sgx_depth_mask; + sh->npot_key = sgx_npot_mask; + /* And the texel classes, split into what the backend takes. */ + if (sgx_cls_units_n > sh->ncls_key) { + unsigned char *p = realloc(sh->cls_key, sgx_cls_units_n); + + if (!p) + return SGX_SHADER_TRANSLATE_FAILED; + sh->cls_key = p; + } + sh->ncls_key = sgx_cls_units_n; + if (sgx_cls_units_n) { + memcpy(sh->cls_key, sgx_cls_units_arr, sgx_cls_units_n); + tex_class = malloc(sgx_cls_units_n); + tex_chunks = malloc(sgx_cls_units_n); + if (!tex_class || !tex_chunks) { + free(tex_class); + free(tex_chunks); + return SGX_SHADER_TRANSLATE_FAILED; + } + for (i = 0; i < sgx_cls_units_n; i++) { + tex_class[i] = (unsigned char) + SGX_TEXKEY_CLASS(sgx_cls_units_arr[i]); + tex_chunks[i] = (unsigned char) + SGX_TEXKEY_CHUNKS(sgx_cls_units_arr[i]); + if (sgx_cls_units_arr[i]) + wide_texel = 1; + } + } + /* Record the swizzles this form was built for, so a later change of + * view is noticed. */ + if (sgx_swz_units_n > sh->nswz_key) { + uint16_t *p = realloc(sh->swz_key, + sgx_swz_units_n * sizeof *p); + + if (!p) + return SGX_SHADER_TRANSLATE_FAILED; + sh->swz_key = p; + } + if (sgx_swz_units_n) { + memcpy(sh->swz_key, sgx_swz_units_arr, + sgx_swz_units_n * sizeof *sh->swz_key); + sh->nswz_key = sgx_swz_units_n; + } else { + sh->nswz_key = 0; + } + sh->stage = stage; + sh->coord_split_in = -1; + sh->coord_split_set = -1; + + s = uir_shader_new((enum uir_stage)stage); + if (!s) + return SGX_SHADER_TRANSLATE_FAILED; + + xs = tgsi_to_uir(s, insns, n, &at); + sh->face_in = -1; + sh->face_key = (unsigned)sgx_face_swap; + if (xs == TGSI_XLAT_OK && stage == UIR_STAGE_FRAGMENT && io) { + unsigned q; + + for (q = 0; q < io->nin && q < 16; q++) + if (io->in_semantic[q] == SGX_SEM_FACE) { + if (sgx_face_apply(s, q, sgx_face_swap)) { + uir_shader_free(s); + return SGX_SHADER_TRANSLATE_FAILED; + } + sh->face_in = (int)q; + break; + } + } + if (xs == TGSI_XLAT_OK && stage == UIR_STAGE_VERTEX && io && + sgx_vs_drop_outputs(s, io)) { + uir_shader_free(s); + return SGX_SHADER_TRANSLATE_FAILED; + } + /* Before the swizzle, which gathers channels: the depth is spread over + * three of them until this has put it back together, and a swizzle + * that names one would keep a single byte. */ + if (xs == TGSI_XLAT_OK && stage == UIR_STAGE_FRAGMENT && + sgx_depth_mask && sgx_depth_expand(s, sgx_depth_mask)) { + if (err) + snprintf(err, errn, "a depth sample does not write a " + "register the expansion can rewrite"); + uir_shader_free(s); + return SGX_SHADER_TRANSLATE_FAILED; + } + /* Before the sRGB conversion, so that decoding sees the channel the + * texture actually holds rather than a constant the swizzle put there. */ + if (xs == TGSI_XLAT_OK && stage == UIR_STAGE_FRAGMENT && + sgx_swz_units_n && + sgx_swz_apply(s, sgx_swz_units_arr, sgx_swz_units_n)) { + if (err) + snprintf(err, errn, "a swizzled sample does not write " + "a register the swizzle can rewrite"); + uir_shader_free(s); + return SGX_SHADER_TRANSLATE_FAILED; + } + if (xs == TGSI_XLAT_OK && stage == UIR_STAGE_FRAGMENT && + sgx_npot_mask) { + npot_sites = sgx_npot_wrap(s, sgx_npot_mask); + if (npot_sites < 0) { + if (err) + snprintf(err, errn, "a sample of a " + "non-power-of-two texture could not " + "be given its own wrap"); + uir_shader_free(s); + return SGX_SHADER_TRANSLATE_FAILED; + } + } + if (xs == TGSI_XLAT_OK && stage == UIR_STAGE_FRAGMENT && + sgx_srgb_mask && sgx_srgb_decode(s, sgx_srgb_mask)) { + if (err) + snprintf(err, errn, "an sRGB sample does not write a " + "register the conversion can rewrite"); + uir_shader_free(s); + return SGX_SHADER_TRANSLATE_FAILED; + } + if (xs != TGSI_XLAT_OK) { + if (err) + snprintf(err, errn, "instruction %u (opcode %u): %s", at, + at < n ? insns[at].opcode : 0u, + tgsi_xlat_status_name(xs)); + uir_shader_free(s); + free(tex_class); + free(tex_chunks); + return SGX_SHADER_TRANSLATE_FAILED; + } + { + size_t i, j; + + for (i = 0; i < s->ninsns; i++) + for (j = 0; j < s->insns[i].nsrc; j++) + if (s->insns[i].src[j].cls == UIR_REG_SAMPLER && + s->insns[i].src[j].index < 32) + sh->sampler_mask |= 1u << + s->insns[i].src[j].index; + } + + for (i = 0; i < nbindings; i++) + uir_add_binding(s, &bindings[i]); + + memset(&opts, 0, sizeof(opts)); + /* Unset. Zero is a legal texel base - see uir_codegen_opts. */ + opts.frag_tex_pa_base = -1; + opts.tex_class = tex_class; + opts.tex_chunks = tex_chunks; + opts.ntex_units = sgx_cls_units_n; + /* The frame delivers the interpolated colour as a packed 8888 dword in + * a primary attribute, and the pixel back end reads o0 as that same + * dword, so ending a fragment program is a move. Packing it as four + * floats produced a flat colour on hardware. */ + /* Note the coordinate operands before translation: after it the sample + * is a register read and the coordinate is gone. + * + * The coordinate is rarely the input itself. The state tracker emits + * MOV TEMP, IN[n].xyxw ahead of a TXP to build the projective + * coordinate, so a sample's own operands name a temporary and reading + * only those found no input at all: nothing was marked sampled, the + * frame iterated no coordinate set, and the sample read a register + * that was never written - the tiler then waited for a texel that + * never arrived. Follow the inputs forward into the temporaries they + * reach instead, and take the coordinate's own operand from that. */ + for (i = 0; i < n; i++) { + unsigned k; + + if (insns[i].dst.file == TGSI_F_TEMPORARY && + insns[i].dst.index + 1 > ntemp) + ntemp = insns[i].dst.index + 1; + for (k = 0; k < insns[i].nsrc; k++) + if (insns[i].src[k].file == TGSI_F_TEMPORARY && + insns[i].src[k].index + 1 > ntemp) + ntemp = insns[i].src[k].index + 1; + } + /* Without it the walk below degrades to the sample's own operands, + * which is what the driver did before. */ + temp_in = ntemp ? calloc(ntemp, sizeof *temp_in) : NULL; + + /* Which samplers have already been seen reading each input. */ + memset(tex_smp, 0, sizeof tex_smp); + memset(tex_dsmp, 0, sizeof tex_dsmp); + memset(tex_direct, 0, sizeof tex_direct); + for (i = 0; i < n; i++) { + unsigned k, reach = 0; + + for (k = 0; k < insns[i].nsrc; k++) + reach |= tgsi_src_inputs(&insns[i].src[k], temp_in, + ntemp); + + /* TEX, TXP and TXL, in Mesa's opcode numbering. */ + if (insns[i].opcode == 52 || insns[i].opcode == 54 || + insns[i].opcode == 72) { + has_tex = 1; + for (k = 0; k < insns[i].nsrc; k++) { + unsigned m = tgsi_src_inputs(&insns[i].src[k], + temp_in, ntemp); + unsigned in = 0; + + if (!m) + continue; + while (!(m & (1u << in))) + in++; + sh->tex_inputs |= 1u << in; + /* Two samplers may read one coordinate set, + * and each still needs its own issue and its + * own texel register, so this counts units + * rather than marking the input. */ + /* Distinct samplers, not sample sites: one + * sampler read from two branches is one unit, + * and counting it twice asked the frame for a + * fourth issue it cannot carry - which took + * alacritty's whole issue list with it. */ + if (in < 16) { + unsigned q, smp = 32; + + for (q = 0; q < insns[i].nsrc; q++) + if (insns[i].src[q].file == + TGSI_F_SAMPLER) { + smp = insns[i].src[q]. + index; + break; + } + if (smp < 32) { + if (!(tex_smp[in] & + (1u << smp)) && + sh->tex_units[in] < 255) { + tex_smp[in] |= + 1u << smp; + sh->tex_units[in]++; + } + } else if (sh->tex_units[in] < 255) + sh->tex_units[in]++; + /* A coordinate the program computed + * takes no iterator of its own, so it + * does not make the input a coordinate + * two units share. */ + if (smp < 32 && + insns[i].src[k].file == + TGSI_F_INPUT && + !(tex_dsmp[in] & (1u << smp)) && + tex_direct[in] < 255) { + tex_dsmp[in] |= 1u << smp; + tex_direct[in]++; + } + } + break; + } + } + + if (insns[i].dst.file == TGSI_F_TEMPORARY && temp_in && + insns[i].dst.index < ntemp) + temp_in[insns[i].dst.index] |= reach; + } + free(temp_in); + temp_in = NULL; + + /* A vertex program's uniforms arrive as primary attributes, DMAed in + * by the vertex PDS just past the vertex record - there is no + * secondary PDS program on this stage to fill an sa bank. */ + /* Uniforms as immediates when the driver asks for it, otherwise in the + * primary attribute bank behind the vertex record. */ + if (sgx_hwtcl_limm_uniforms()) + opts.vtx_uniform_limm = 1; + else + opts.vtx_uniform_pa_base = (sgx_hwtcl_sa_uniforms() && + !getenv("SGX_TA_PA_UNI")) ? 0 : + SGX_HWTCL_MAT_AO; + /* The pixel pack has to match whichever record feeds it: the draw + * module's carries the colour blue first, a program the part runs + * writes it as the shader computed. Which one that is, is a per-frame + * decision - a context that falls back mid-run changes it - so it + * cannot be settled here, and keying it on the request rather than on + * the frame reversed every fragment a fallen-back context drew. The + * pack stays on the draw module's order until the fragment program can + * be selected per frame. */ + opts.vtx_no_literal_pool = 1; + if (stage == UIR_STAGE_VERTEX && io && io->nout) { + if (!sgx_vs_out_slots(io, opts.vtx_out_slot, io->nout)) { + uir_shader_free(s); + if (err && errn) + snprintf(err, errn, "an output has no record " + "slot"); + return SGX_SHADER_TRANSLATE_FAILED; + } + opts.vtx_out_slots = 1; + /* The record's width in quads, which is what the MTE emits per + * vertex: the backend fills every slot in it the program does + * not write. Rounded up, so a width that is not a whole number + * of quads still has its last, partly used slot covered. */ + opts.vtx_out_slots_emitted = + (sgx_hwtcl_record_dwords(0) + 3u) / 4u; + if (sgx_vtx_have_dw) { + unsigned q; + + for (q = 0; q < io->nout && q < 16; q++) + opts.vtx_out_dw[q] = sgx_vtx_dw_of( + io->out_semantic[q], io->out_index[q], + opts.vtx_out_slot[q]); + opts.vtx_out_dwords = 1; + } + /* What the MTE emits is decided by the highest slot used, not + * by how many outputs there are: they land at o[4 * slot], and + * a program writing position and one generic leaves slot 1 + * empty between them. Group 10 has to name that size or the + * fragment task waits for dwords the vertex stage never + * sends. */ + { + unsigned i, top = 0; + + for (i = 0; i < io->nout && i < 16; i++) + if (opts.vtx_out_slot[i] > top) + top = opts.vtx_out_slot[i]; + sh->nvtxout = top + 1u; + } + /* Off by default. The mismatch it corrects is real and the two + * paths disagree in the tree, but no workload that reaches the + * part with an iterated colour could be run to confirm the + * corrected order on hardware - ioquake3 never takes the + * hardware path for a coloured draw. */ + if (getenv("SGX_VS_BGRA")) { + unsigned i; + + for (i = 0; i < io->nout && i < 16; i++) + if (opts.vtx_out_slot[i] == 1 && + sgx_vs_colour_bgra(s, i)) { + uir_shader_free(s); + if (err && errn) + snprintf(err, errn, "the " + "colour slot could " + "not be reordered"); + return SGX_SHADER_TRANSLATE_FAILED; + } + } + } + + /* Only a program that samples gets its colour already packed: the + * texture unit hands the texel over as a primary attribute in that + * form, so moving it to the output is the whole program. One that + * computes its colour holds four floats instead and has to be packed + * properly, which is what a constant-coloured program does - and what + * it used to get wrong, writing the bit pattern of a float into the + * pixel. */ + /* Whether the output is already a packed 8888 dword, or four floats + * that still have to be packed. + * + * The frame's iterator hands the interpolated colour over as one + * packed dword in pa0, and the texture unit delivers a texel the same + * way. So a program that only moves one of those to its output is + * already packed - treating it as four floats reads the bit pattern of + * the dword as a float and saturates, which is what made lit geometry + * come out per-fragment black or fully saturated instead of shaded. + * + * A program that *computes* its colour - from a uniform, a constant, + * arithmetic - holds four real floats and must be packed properly. + * Claiming it was already packed sent the bit pattern of a float to + * the pixel back end, which is how a red uniform arrived as 0x80. + * + * So: packed if the program samples, or if it is a pass-through of an + * input to the output. Anything else packs. */ + /* Packed only when nothing computes: a program that samples into a + * temporary and then does arithmetic holds floats, and claiming those + * were already packed sends a float's bit pattern to the pixel. */ + /* A view swizzle is applied after the TGSI was read: its moves permute + * float channels, so the texel cannot go out as the dword it arrived in. + * Nor can a float or 16-bit texel, which is unpacked and packed again. + * The sRGB decode is the same case for the same reason - it is inserted + * after this, and it computes. */ + opts.frag_out_packed = frag_is_passthrough(insns, n) && + !frag_computes_with_tex(insns, n) && + !swz_any_nonidentity() && + !(sgx_srgb_mask & sh->sampler_mask) && + !(sgx_depth_mask & sh->sampler_mask) && + !(wide_texel && sh->sampler_mask); + + /* Where the fragment stage's inputs land. A program that samples, or + * that does nothing but move an input to the output, reads the packed + * colour in pa0 and the texels behind it - the shape every capture + * has. Anything else holds real floats, and a packed dword read as + * four floats is what made a computed varying come out black, so its + * inputs are iterated into coordinate sets instead. */ + /* SGX_FORCE_SMP puts a program the iterator could have served onto the + * shader-issued path, so a case with a known-correct result can + * exercise it. */ + /* A program that samples with a varying and also reads that varying's + * other components cannot be served by the iterator: the unit's texel + * takes a register inside the set, so those components arrive as the + * texel. It samples for itself instead, where the set is an ordinary + * iterated one. Mesa packs alacritty's foreground colour into the + * texture coordinate's z and w, which is exactly this. */ + { + unsigned q; + + io_nin = io ? io->nin : 0u; + for (q = 0; q < io_nin && q < 16; q++) + if (((sh->tex_inputs >> q) & 1) && + sh->tex_units[q] <= 1 && + frag_in_outside(insns, n, q, 1)) + break; + smp_and_read = q < io_nin && q < 16; + } + opts.vtx_pack_colour = stage == UIR_STAGE_VERTEX && sgx_vtx_pack_colour; + /* A wrapped coordinate is computed, so the iterated form cannot serve + * it whatever anything else asks for - this comes first. */ + opts.frag_tex_preiterated = npot_sites ? 0 : + getenv("SGX_NO_SMP") ? 1 : + getenv("SGX_FORCE_SMP") ? 0 : + smp_and_read ? 0 : + sgx_preiter_pref >= 0 ? sgx_preiter_pref : + frag_tex_is_iterated(insns, n); + /* The caller asks for the iterated form first and falls back only + * when it fails, so a sample the iterator cannot serve has to fail + * it, not merely not prefer it. A volume or a cube is one: the + * iterator hands the unit two coordinates and a projection on this + * core - measured, the slice never arrived under TEXPROJ_T or RHW, + * and TEXPROJ_S is reserved here (sgxdefs.h:3695) - so the sample + * goes to the USE, which reads all three. Keyed on the sampler's + * target, never on how many components the coordinate has: a 2D + * TXP carries three and the iterator applies its divide. SGX_NO_SMP + * still forces the iterated form, for the reserved-code probe. */ + if (stage == UIR_STAGE_FRAGMENT && opts.frag_tex_preiterated && + !getenv("SGX_NO_SMP") && frag_has_three_coord(insns, n)) { + if (err && errn) + snprintf(err, errn, "a volume or cube sample: the " + "iterator delivers two coordinates on this " + "core"); + uir_shader_free(s); + return SGX_SHADER_TRANSLATE_FAILED; + } + /* One varying can carry two coordinate pairs - glamor's composite is + * source and mask, and mesa packs the two vec2s into one vec4. The + * varying is served as one ordinary set and both units sample with it. + * + * A sampled set is sampled with its first two floats, so one set can + * only offer one pair; the two units therefore both read pair .xy. + * That is exact whenever the two pairs hold the same coordinate, which + * is the composite every caller measured here builds, and it is what + * the remaining two shapes fail to do at all: a second set + * manufactured from the same varying, with the record moving each pair + * to its front, reads its mask as opaque at every pixel - texel (0,0), + * a unit that got no coordinate - and refusing the iterated form so + * the USE samples for itself does the same. Measured on "composite + * varyings", which passes this way and fails both others, with the + * isolated suite at 164 of 170 against 163 and nothing moving back. + * + * Two genuinely different pairs are still open: sgx_fs_set_coord() + * carries the per-set pair the record would need and the vertex + * buffer already honours it, but nothing yet computes anything but + * 0,1, because both paths that could use it are the two that fail. */ + + if (stage == UIR_STAGE_FRAGMENT && io) { + unsigned colour = 0, set_base[XPSB_NSET_MAX] = { 0 }; + unsigned tex_base[XPSB_MAX_TEX] = { 0 }, i, ntex = 0, nvary = 0; + unsigned ncoord = 0; + int v0_taken = 0; + unsigned pass; + /* The set a volume's sample must read its coordinate from, + * and the input it belongs to. That set is an ordinary + * iterated varying rather than the sampled one, so the + * mapping below cannot find it by set_varying - the sampled + * set names the same input and comes first. -1 for every + * program that is not a volume or a cube. */ + int vol_iter_set = -1, vol_iter_in = -1; + /* SGX_NO_PACKED_COLOUR carries a pass-through colour as an + * ordinary coordinate set instead of the record's packed + * colour dword, which is the shape a shader-written varying + * takes and the one measured to run on the part. */ + /* And a fixed-function draw always does, which is + * FIX_HW_BRN_25211: the erratum is on SGX535 rev 1.2.1's list + * (hwdefs/sgxerrata.h:785) and the vendor's workaround takes + * the vertex colour off the MTE's base slot and delivers it + * through a texture coordinate set - EURASIA_MTE_BASE is not + * set and EURASIA_MTE_TEXDIM_UVST goes in the coordinate + * selects instead (opengles1/fftnlgles.c:900-975), the + * iterator names that set rather than USEISSUE_V0 + * (opengles1/usegles.c:2077-2118), and the colour is clamped + * to [0,1] because that path does not clamp where the base + * colour's iterator did (codegen.c:1369-1446). The clamp is + * the fragment output's saturating U8 pack here, which is + * where this driver's colour ends up. + * + * This is the shape a fixed-function colour pass-through + * stopped the tiler in: nine frames in ten lost, no MMU + * fault, with the record's colour already four F32 as the MTE + * reads it. + * + * Opt-in, because routing it costs more than the stall it is + * meant to avoid: measured on the part, with the colour on a + * coordinate set the triangle and the five packed-colour cases + * all render the clear colour and the line strip draws nothing + * at all, where every one of them passes on the base slot. + * SGX_FF_COLOUR_SET=1 turns it on. */ + int ff_colour_set = getenv("SGX_FF_COLOUR_SET") && + !sh->tex_inputs; + int in_packed = !getenv("SGX_NO_PACKED_COLOUR") && + !ff_colour_set && + frag_in_is_packed_colour(insns, n); + /* A varying the program computes with can come over packed as + * well, and be unpacked at entry. Iterating one as a + * coordinate set instead is what the frame spends its time on: + * the same scene renders in 5 ms delivered packed and 106 + * iterated. Only a varying nothing samples - a coordinate has + * to be a set, that is how the unit reads it. */ + /* Off by default: the packed form is four bytes, so a varying + * carrying anything but a colour in [0,1] loses range and + * precision to it, and which varyings those are is not + * something the program says. A pass-through is safe because + * the value was already an 8888 colour on its way to the + * output, which is what in_packed covers. */ + /* And on when the sets would not otherwise go round. The + * frame carries XPSB_NSET_MAX coordinate sets and a sampled + * varying has to have one - only a coordinate set reaches a + * texture unit - so when a program has more varyings than + * that, the ones nothing samples come over packed instead. + * That is the shape the captured two-unit frame has: its + * iterated colour rides on useissue 10, the packed colour, + * while both coordinate sets go to the two units. Numbering + * a colour into a third set instead asks the frame for a + * set index no capture exercises, and nothing is delivered - + * ioquake3's lit surfaces came out black for it. */ + int nvarying = (int)(io->nin < 16 ? io->nin : 16); + /* A varying that shares the program with a sampled one is + * packed too, not only one past the sets. Measured on + * ioquake3's menu: its text is a glyph coordinate and a + * colour, the colour took a set of its own and never arrived - + * 180 lit pixels where the packed path gives 13279 and the + * menu reads correctly. SGX_NO_SMP_PACK puts it back. */ + /* Only when it is the one varying nothing samples: there is a + * single colour register, and packing the first of two took it + * from the one that had no set left to go in. + * + * And only a colour. The packed path is an 8888 dword clamped + * to zero..one, which is what a colour is and what a + * coordinate is not: packing ioquake3's animated logo + * coordinate quantised it and the flame came out scaled over + * the whole screen instead of its banner. */ + int has_sampled = sh->tex_inputs != 0; + int nunsampled = 0, ncolour = 0, last_sampled = -1, uq; + int nface = 0; + + for (uq = 0; uq < nvarying; uq++) + if (io->in_semantic[uq] == SGX_SEM_FACE) { + nface++; + } else if (!((sh->tex_inputs >> uq) & 1)) { + nunsampled++; + if (io->in_semantic[uq] == SGX_SEM_COLOR) + ncolour++; + } else { + last_sampled = uq; + } + /* A colour beside a sampled input. + * + * Restricted to one the sampled inputs are declared before, + * and that restriction is a refusal rather than a rule: it + * says where the packed colour is known to work, not where it + * ought to. Dropping it renders ioquake3's world at about + * 4500 colours a frame against 50000, stable, from a clean + * boot with no recoveries - so the remaining defect is real + * and is not the declaration order, which every step below + * now settles by semantic. SGX_NO_SMP_PACK turns the packing + * off entirely. */ + int beside = has_sampled && nunsampled == 1 && ncolour == 1 && + last_sampled >= 0 && last_sampled < nvarying - 1 && + !getenv("SGX_NO_SMP_PACK"); + int unpack_in = !in_packed && + (getenv("SGX_IN_UNPACK") != NULL || + ((nvarying - nface > XPSB_NSET_MAX || beside) && + !getenv("SGX_NO_AUTO_UNPACK"))); + + { + unsigned q; + + sh->nin = io->nin < 16 ? io->nin : 16; + for (q = 0; q < sh->nin; q++) { + sh->in_sem[q] = io->in_semantic[q]; + sh->in_sem_idx[q] = io->in_index[q]; + } + } + memset(&sh->attribs, 0, sizeof sh->attribs); + sh->colour_passthrough = in_packed; + sh->packed_in = -1; + sh->fragcoord_set = -1; + sh->coord_split_in = -1; + sh->coord_split_set = -1; + if (getenv("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: fs inputs: %u, tex_inputs 0x%x, " + "packed colour %d\n", io->nin, + (unsigned)sh->tex_inputs, in_packed); + /* One set per input, in declaration order: a sampled + * coordinate and an iterated colour are different quantities + * and cannot share a set, which counting them separately made + * them do. */ + /* Sampled inputs first, so that the coordinate sets take the + * low set numbers. The set index is the texissue, and the + * captured two-unit frame samples sets 0 and 1; a coordinate + * that an iterated colour pushed to set 2 asked the unit for + * a coordinate set no capture has, and came back black. */ + for (pass = 0; pass < 2; pass++) + for (i = 0; i < io->nin && i < 16; i++) { + if (!((sh->tex_inputs >> i) & 1) != !!pass) + continue; + /* gl_FrontFacing is computed, not iterated. */ + if (io->in_semantic[i] == SGX_SEM_FACE) + continue; + /* By register number, which is what the loop counts + * and what tex_inputs and tex_units are keyed by. + * in_index[] is indexed by register but holds the + * *semantic* index, so shifting by it tested the + * wrong bit for every program whose sampled varying + * is not its first: "IN[0] COLOR, IN[1] TEXCOORD[0]" + * marked neither input sampled, and the draw came out + * untextured. */ + int sampled = (sh->tex_inputs >> i) & 1; + /* Only the iterator's own fetch makes the set a + * texture unit's. A program that samples for itself + * needs the coordinate iterated and nothing else: + * marking the set sampled issued the unit as well, so + * the register the program read held the texel the + * frame had fetched and the sample used its bit + * pattern as a coordinate. */ + int issue = sampled && opts.frag_tex_preiterated; + /* Two units reading one set need an issue each, but a + * program that samples for itself reads that set's + * registers as its own values, and a second unit + * marked sampled writes its texel over them. They get + * their issues as units instead. */ + int smp_multi = !opts.frag_tex_preiterated && + tex_direct[i] > 1; + unsigned set = nvary; + /* Past the coordinate sets a varying rides the V0 + * colour iterator: not a coordinate set, and carried + * in the record's colour quad as four whole floats. + * The packed colour register cannot hold one - it is + * an 8888 dword clamped to zero..one, which is what a + * colour is and what a position is not. glmark2's + * terrain packs five varyings into four slots and the + * fourth holds a view-space position, hundreds of + * units that saturated to one and left every surface + * lit by ambient alone. */ + int on_v0 = !sampled && ncoord >= XPSB_NCOORD_MAX && + !v0_taken && nvary < XPSB_NSET_MAX && + !getenv("SGX_NO_V0_VARY"); + + if (!sampled && in_packed) { + /* Recorded here too, or the record's colour + * slot is chosen by a different rule than the + * one the program was compiled against. */ + /* And the backend told, as the branch below + * tells it. packed_in alone makes the vertex + * program pack the colour into the record's + * one colour dword while the fragment program + * is still compiled to read four iterated + * ones, so the fragment task waited on three + * the frame never issued: under the part's + * transform every frame of a fixed-function + * draw was lost, tiling engine stopped and no + * MMU fault. Both halves say packed now. */ + opts.frag_in_packed |= 1u << i; + if (sh->packed_in < 0) + sh->packed_in = (int)i; + continue; /* it is the packed colour */ + } + /* Packed and unpacked at entry, so it costs the + * colour register rather than a set. */ + /* Only the varying that does not fit. There is one + * colour register, so packing every unsampled input + * gave them all the same base and they read each + * other's value; the ones that fit keep sets of their + * own. SGX_IN_UNPACK still packs every one, which is + * the speed experiment this began as. */ + /* By the sets already spoken for, not by the input's + * own number. Sampled inputs are numbered first, so + * by the time an iterated one is reached every + * coordinate has its set - and a program whose + * coordinates come after its colour was refused for + * "3 varyings" when three is exactly what the frame + * carries. ioquake3's lit surfaces are that shape. + * Only one can be packed: there is one colour + * register. */ + if (!sampled && unpack_in && sh->packed_in < 0 && + (getenv("SGX_IN_UNPACK") || + (beside && io->in_semantic[i] == SGX_SEM_COLOR && + (int)i > last_sampled) || + (!on_v0 && ncoord >= XPSB_NCOORD_MAX && + io->in_semantic[i] == SGX_SEM_COLOR))) { + opts.frag_in_packed |= 1u << i; + if (sh->packed_in < 0) + sh->packed_in = (int)i; + continue; + } + /* The frame describes two coordinate sets. A third + * varying used to be given set 0's base and read the + * wrong register, which is a shader that renders + * wrongly with nothing said - so it is refused. */ + if (set >= XPSB_NSET_MAX || + (!on_v0 && ncoord >= XPSB_NCOORD_MAX)) { + if (err && errn) + snprintf(err, errn, "%u varyings, and " + "the frame carries %u " + "coordinate sets and one " + "packed colour", io->nin, + (unsigned)XPSB_NSET_MAX); + uir_shader_free(s); + return SGX_SHADER_TRANSLATE_FAILED; + } + nvary++; + if (on_v0) { + sh->attribs.set[set].on_colour = 1; + v0_taken = 1; + } else + ncoord++; + /* Three registers is the texture unit's own coordinate + * form. A varying the program samples with itself is + * an ordinary four-component one - a projective + * sample divides by its w, and iterating only three + * left that read off the end of the set. */ + { + unsigned char c0 = 0, c1 = 1; + /* Off by default: it is not right yet. Built, and the + * encoding it produces is correct - g14 names a + * three-float projected set and a four-float one - but + * the texel never reaches the program. Sweeping + * SGX_TEX_PA over the whole bank with the iterated set + * first finds the varying at 0..2 and nothing at 3..5; + * with the sampled set first nothing renders at all. + * Where the primary emitter puts a texel for a sampled + * set that shares a frame with an iterated one is not + * established, and xpsb_attrib_bases() models it as if + * it followed in issue order, which these two sweeps + * say it does not. SGX_COORD_SPLIT=1 to carry it on. */ + /* Off: the frame cannot express it. A varying the unit + * samples and the program reads needs a coordinate, a + * divisor and a value out of one set. The unit wants a + * three-component coordinate - texdim 2 stalls the + * render - so the third float is the divisor, and a + * set four wide delivers only three registers to the + * program, so the fourth never arrives. Two sets do + * not work either: with the iterated one first the + * texel appears nowhere in the bank, with the sampled + * one first nothing renders. Each of those was built + * and measured; SGX_COORD_SPLIT=1 to take it up. */ + int split = issue && frag_input_used(insns, n, i) && + getenv("SGX_COORD_SPLIT") && + frag_coord_comps(insns, n, i, &c0, &c1); + + /* A varying the unit samples and the program also + * reads cannot be one set: a sampled set is three + * floats projected, so its third component is the + * divisor, and a scalar Mesa packed in there would be + * divided into the coordinate. It gets two - an + * iterated one carrying the varying as it stands, + * which the program's reads are based on, and a + * sampled one behind it that the record fills with the + * coordinate and a divisor of one. + * + * The iterated one comes first because that is the + * order the primary emitter lays the bank out in: the + * iterated values, then the texels. */ + if (split) { + /* One set, four floats, both sampled and + * iterated: the unit takes the coordinate off + * the front and the program is handed the same + * four registers, so the component it reads + * arrives with them. Three floats cannot do + * this - that set is projected, and the + * divisor has the third. */ + sh->coord_split_in = (int)i; + sh->coord_split_set = (int)set; + sh->attribs.set[set].width = 4; + sh->attribs.set[set].sampled = sh->tex_units[i]; + sh->attribs.set[set].iterated = 1; + } else { + /* SGX_SET_ALIGN4 declares every set four floats + * wide, to tell a record packed by width apart + * from one on fixed four-dword slots. Measured on + * ioquake3: it does not align them, it shifts the + * iterated colour a second register - the world + * goes from blue-dominant to green-only, the same + * picture SGX_ITER_PROJ produces. A recorded + * negative; the sets are packed by width. */ + /* Two floats for a set the unit iterates: + * the coordinate and nothing else. The third + * was only ever the TEXPROJ_T divisor, and + * the issue asks for TEXPROJ_RHW now, so the + * projection comes off the position's own + * reciprocal-w plane. It is also the third + * float the hardware-transform path never + * filled - there the vertex program writes + * the varying's own z there, not a one. + * SGX_SET_W3 keeps the three-float form. */ + sh->attribs.set[set].width = + (issue && !getenv("SGX_SET_ALIGN4")) ? + (getenv("SGX_SET_W3") ? 3 : 2) : 4; + /* A four-float iterated set renders the same + * geometry eighteen times slower than a frame + * that also carries a three-float one, so how + * wide the set is declared is worth sweeping + * against the clock. */ + { + const char *we = getenv("SGX_SET_W"); + + if (we && *we && !issue) { + unsigned wv = (unsigned) + strtoul(we, NULL, 0); + + if (wv >= 2 && wv <= 4) + sh->attribs.set[set].width = + (uint8_t)wv; + } + } + if (issue) + sh->attribs.set[set].sampled = + sh->tex_units[i]; + /* A set the program samples for itself still + * carries a texture coordinate, and the unit + * is what delivers one: SMP reads the + * coordinate the iterator produced, so a set + * issued only as a varying leaves the sampler + * at its default and every pixel comes back + * the same texel. The old switch for this + * marked set 0 whichever set the program + * actually sampled, which for a program with a + * colour varying ahead of its coordinate is + * the wrong one. */ + /* But not when the program reads the set's + * other components as values. The unit's + * texel takes a register inside the set, so + * what the program reads past the coordinate + * is the texel and not its varying: alacritty + * keeps its foreground colour in the + * coordinate's z and w, and its glyphs came + * out magenta for it. A program that samples + * for itself needs no unit anyway. */ + /* And only when one sampler reads the set. Two + * units on one coordinate need an issue each, + * and taking those away stalls the render - + * "composite varyings" and "a8 mask, two + * varyings" went black for it. On the + * shader-issued path the issue is a unit, not + * a sampled set: see smp_multi. */ + else if (sampled && !smp_multi && + !getenv("SGX_NO_SMP_TCSET") && + (tex_direct[i] > 1 || + getenv("SGX_NO_VALUE_GATE") || + !frag_in_outside(insns, n, i, + sampled))) + sh->attribs.set[set].sampled = + sh->tex_units[i]; + if (tex_direct[i] <= 1 && + !getenv("SGX_NO_VALUE_GATE") && + frag_in_outside(insns, n, i, sampled)) + sh->attribs.set[set].values = 1; + /* The units still need their issues even when + * the set is not described as sampled. Kept + * apart from `sampled`, which is what takes a + * float out of the set and narrows the record: + * a program that computes its own coordinate + * reads the set's other components as values + * and must keep them, but every unit reading + * it wants the texture state its DOUTT + * writes. Without this a program with four + * such units stalled the core - no MMU fault, + * nothing issued for three of them. */ + if (!sh->attribs.set[set].sampled) + sh->attribs.set[set].units = + sh->tex_units[i]; + if (getenv("SGX_DEBUG")) + fprintf(sgx_log(), "sgx: set %u in %u:" + " issue %d sampled %d units %u/%u" + " outside %d -> sampled %u" + " values %u\n", set, i, + issue ? 1 : 0, sampled ? 1 : 0, + sh->tex_units[i], + tex_direct[i], + frag_in_outside(insns, n, i, + sampled), + sh->attribs.set[set].sampled, + sh->attribs.set[set].values); + /* A program that reads past the coordinate + * needs those registers in the set, so it + * keeps the floats it had. */ + if (sh->attribs.set[set].values && + sh->attribs.set[set].width < 3) + sh->attribs.set[set].width = 3; + sh->attribs.set[set].iterated |= + (uint8_t)!issue; + /* Which set carries the vertex colour under + * the FIX_HW_BRN_25211 routing: the iterator + * flat-shades from that set, since the MTE's + * shade model only selects a base colour and + * this colour is not on it. */ + if (ff_colour_set && !issue && + !sh->attribs.colour_set && + io->in_semantic[i] == SGX_SEM_COLOR) + sh->attribs.colour_set = set + 1u; + /* The record carries the coordinate and then + * the divisor the iterator applies, so it is + * four floats wide however the set is read. */ + /* Off by default. Mesa's fixed-function shader + * attaches a projector to every sample, so + * every ioquake3 draw is a TXP - but its + * projector is 1.0, and sgx_vbuf_sampled_- + * divisor() already skips a divide by one. All + * the marking did was widen the set to four + * floats, and with that the texel never + * reached the program: the whole level drew + * untextured. Off, it is textured and the + * frame is 12 fps where it was 1. + * SGX_ITER_PROJ puts it back for a caller with + * a genuinely projective coordinate. */ + if (frag_input_projected(insns, n, i) && + getenv("SGX_ITER_PROJ")) { + sh->attribs.set[set].projected = 1; + sh->attribs.set[set].width = 4; + } + /* A volume's coordinate is three floats and + * all three are the coordinate, so the set + * is declared four wide - UVST - and the + * program reads the first three registers. + * + * Not three: measured on the booklet + * (work/feat-texunit/hw4), a sampled set + * declared three floats delivered the + * coordinate pair to the USE and not the + * third component. u and v arrived, the + * slice never did, and every draw sampled + * one slice - though the record carried 33 + * distinct third floats and the issue asked + * for USEDIM_3D. A fourth float puts the + * slice where the iterator has no reason to + * drop it, whether or not it takes a sampled + * set's last component as the projector it + * would be in a UVT set. SGX_3D_SETW=3 is + * the three-float form. */ + if (frag_input_volume(insns, n, i) && + !sh->attribs.set[set].projected) { + const char *sw = getenv("SGX_3D_SETW"); + + sh->attribs.set[set].width = + (sw && *sw == '3') ? 3 : 4; + sh->attribs.set[set].volume = 1; + } + /* And the coordinate the program will read. + * + * Three shapes of sampled set have now been + * measured on the part and none of them + * delivers a third component to the USE: + * three floats as UVS, four as UVST, and the + * three-float form as a control - all three + * gave u and v and one constant slice + * (work/feat-texunit/hw, hw2, hw3, hw4). + * Since the four- and three-float forms are + * indistinguishable in the result, the + * iterator is not truncating at a declared + * width: a sampled set carries the TAG's + * coordinate pair and nothing more. + * + * The one construction on this part measured + * to deliver more than two components is an + * ordinary iterated varying - glmark2's + * untextured scenes iterate four apiece - so + * the coordinate takes a set of its own, + * iterated and not sampled, fed from the same + * varying. It gets its own DOUTI issue with + * texissue NONE, and the sample reads its + * registers. The sampled set stays, and stays + * first, so the texture issue still leads the + * primary program the way every capture has + * it. SGX_3D_COORD=samp goes back to reading + * the sampled set's registers. */ + if (sh->attribs.set[set].volume && issue && + set + 1u < XPSB_NSET_MAX && + !getenv("SGX_NO_3D_ITERSET")) { + const char *cs = getenv("SGX_3D_COORD"); + + set++; + nvary++; + sh->attribs.set[set] = + sh->attribs.set[set - 1]; + sh->attribs.set[set].sampled = 0; + sh->attribs.set[set].iterated = 1; + sh->set_varying[set] = (uint8_t)i; + sh->set_c0[set] = 0; + sh->set_c1[set] = 1; + sh->attribs.nset = set + 1; + if (!(cs && !strcmp(cs, "samp"))) { + vol_iter_set = (int)set; + vol_iter_in = (int)i; + } + } + } + } + /* gl_FragCoord: an input with the position semantic, + * which no vertex program writes as a varying. The + * record already carries the window position the + * epilogue computed, so the set is fed from that. */ + if (io->in_semantic[i] == SGX_SEM_POSITION) + sh->fragcoord_set = (int)set; + if (set < XPSB_NSET_MAX) { + sh->set_varying[set] = (uint8_t)i; + sh->set_c0[set] = 0; + sh->set_c1[set] = 1; + } + sh->attribs.nset = set + 1; + /* Two samplers reading one varying at different + * components - a source at .xy and a mask at .zw, + * which is what Mesa's packing makes of every glyph + * composite. One set cannot serve both: a set is what + * a unit samples with. So the second coordinate takes + * a set of its own, fed from the same varying, and + * the record moves its components to the front. + * + * Without this the second sample kept a computed + * coordinate, the iterated path refused the program + * for it, and the shader-issued path refuses two + * units - so the composite did not build at all and + * the X server drew its text in software. */ + /* One varying carrying two coordinate pairs is served + * as one set, sampled by both units. A sampled set is + * sampled with its first two floats, so both read the + * .xy pair - exact while the two pairs carry the same + * coordinate, which every composite measured here + * does. A second set manufactured from the same + * varying was tried instead and reads its mask as + * opaque everywhere - texel (0,0), a unit that got no + * coordinate - and so did sending the sample to the + * USE. Measured on "composite varyings", which passes + * this way and fails both others; isolated suite 164 + * of 170 against 163. */ + (void)ntex; + } + /* A program that samples for itself gets no DOUTT under the + * rule recovered from the vendor - its descriptor array is + * left empty. Whether the texture unit still has to be issued + * for the sample to complete was inferred from a capture + * rather than measured, so this keeps the issue while that is + * settled on hardware. */ + if (!opts.frag_tex_preiterated && getenv("SGX_SMP_DOUTT")) { + unsigned k, ns = 0; + + for (k = 0; k < n; k++) + if ((insns[k].opcode == 52 || + insns[k].opcode == 54 || + insns[k].opcode == 72) && + insns[k].nsrc > 1 && + insns[k].src[1].file == TGSI_F_SAMPLER && + insns[k].src[1].index + 1 > ns) + ns = insns[k].src[1].index + 1; + if (ns && sh->attribs.nset) + sh->attribs.set[0].sampled = (uint8_t)ns; + } + /* Every captured frame issues its texture first. A frame whose + * iterated varying comes ahead of its coordinate puts the + * DOUTT on the second issue, and that is the one arrangement + * left untested for why such a frame samples nothing. */ + if (getenv("SGX_SET_TEX_FIRST") && sh->attribs.nset == 2 && + sh->attribs.set[1].sampled && !sh->attribs.set[0].sampled) { + struct xpsb_attribs sw = sh->attribs; + + sh->attribs.set[0] = sw.set[1]; + sh->attribs.set[1] = sw.set[0]; + sh->sets_swapped = 1; + } + /* A pass-through whose colour arrives packed is unpacked at entry - + * every packed input is, because a computing program needs the four + * bytes as floats - so the register the tail would forward holds a + * float, not the dword the pixel back end reads. Two changes that are + * each right alone: the packed tail is c01da8b (26 Aug), the entry + * unpack 0cd2838 (28 Aug), and they have contradicted each other + * since. Let such a program pack its result like any other; the pack + * path is the one every rendering program already uses. + * + * Forwarding the packed dword instead reaches the back end in the + * iterator's byte order, which is a red/blue swap - measured on + * tricol, which wants ff0000 and read 0012ed. */ + if (opts.frag_in_packed && opts.frag_out_packed) { + opts.frag_out_packed = 0; + /* And reversed, because the record's colour reaches the entry + * unpack in the opposite byte order to the one the pack path + * writes for the target. A program that computes is unaffected: + * it never forwarded the record's dword, and solid colours, + * which come from a uniform rather than the record, are + * already right. */ + opts.frag_out_reverse = 1; + } + /* Half-float delivery for the sets that can take it. The + * iterator's USEFORMAT field says F16 and the set arrives in + * half as many primary attribute registers - four components + * in two - which is what the pixel data-master word sizes a + * task from, so it is the same lever the packed colour is + * without the packed colour's eight bits a channel. + * + * Only a set nothing samples. A sampled set's coordinate goes + * to the TAG, whose own format is the texture descriptor's, + * and the register a texture issue puts its texel in sits + * inside the set - so narrowing the set would move it. An + * iterated set has no such passenger. + * + * Off by default: the encoding is the DDK's and the register + * accounting follows it, but neither has run on the part, and + * a varying that arrives at the wrong width is a wrong picture + * with nothing said. SGX_F16_VARY=1 turns it on. */ + if (getenv("SGX_F16_VARY")) { + unsigned q; + + for (q = 0; q < sh->attribs.nset && + q < XPSB_NSET_MAX; q++) { + unsigned in = sh->set_varying[q]; + + if (sh->attribs.set[q].sampled || + !sh->attribs.set[q].iterated) + continue; + if (in >= 16 || + (opts.frag_in_packed & (1u << in))) + continue; + sh->attribs.set[q].f16 = 1; + opts.frag_in_f16 |= 1u << in; + } + } + sh->attribs.colour = in_packed || opts.frag_in_packed ? 1u : 0u; + xpsb_attrib_bases(&sh->attribs, &colour, set_base, tex_base); + if (getenv("SGX_DEBUG")) { + unsigned q; + + fprintf(sgx_log(), "sgx: bases: nin %u texin 0x%x " + "packed 0x%x f16 0x%x colour %u", io->nin, + (unsigned)sh->tex_inputs, + opts.frag_in_packed, opts.frag_in_f16, colour); + for (q = 0; q < sh->attribs.nset && + q < XPSB_NSET_MAX; q++) + fprintf(sgx_log(), ", set %u at %u (w %u samp %u " + "iter %u) tex at %u", q, set_base[q], + sh->attribs.set[q].width, + sh->attribs.set[q].sampled, + sh->attribs.set[q].iterated, + tex_base[q]); + fprintf(sgx_log(), "\n"); + } + ntex = nvary = 0; + for (i = 0; i < io->nin && i < 16; i++) { + int sampled = (sh->tex_inputs >> i) & 1; + + if (!sampled && (in_packed || + (opts.frag_in_packed & (1u << i)))) { + opts.frag_in_base[i] = (unsigned char)colour; + continue; + } + /* A sampled input's own register holds the coordinate + * the set iterates; what the program reads for the + * sample is the texel, which sits behind it. */ + /* By the set this input was actually given, not by + * the order the inputs are declared in. Sampled sets + * are numbered first, so the two are no longer the + * same, and reading set_base by declaration order + * handed a program its coordinate's registers where + * it asked for its colour. */ + { + unsigned q, st = nvary; + + for (q = 0; q < sh->attribs.nset && + q < XPSB_NSET_MAX; q++) + if (sh->set_varying[q] == (uint8_t)i) { + st = q; + break; + } + if (sh->sets_swapped) + st ^= 1u; + /* A volume's coordinate is the iterated set, + * not the sampled one that names the same + * varying and comes first. */ + if (vol_iter_in == (int)i && vol_iter_set >= 0) + st = (unsigned)vol_iter_set; + opts.frag_in_base[i] = (unsigned char) + set_base[st < XPSB_NSET_MAX ? st : 0]; + } + if (sampled && ntex < XPSB_NSET_MAX) + ntex++; + if (nvary < XPSB_NSET_MAX) + nvary++; + } + opts.frag_in_bases = 1; + opts.frag_tex_pa_base = (int)tex_base[0]; + } + /* The register allocator's budget, not the hardware's. Sixty-four was + * enough while a program had to fit eight instructions; a real one + * needs more, and what the hardware can be told about is a separate + * question - see sgx_temp_field(). */ + opts.max_temps = SGX_SHADER_MAX_TEMPS; + /* The allocator spends registers it does not have to: a three + * instruction texture shader comes out with ten temporaries, and temps + * per thread is what bounds how many pixel threads the part runs at + * once. Squeezing the budget does not buy that back, though - it is + * not a lever. Every value low enough to change the allocation (12 and + * below, for glmark2's cube) gets the client killed rather than a + * failed link, so the render times it appears to produce are dying + * runs, not faster ones. The kill is worth a look on its own: the + * codegen grow loop is bounded at SGX_SHADER_MAX_CODE, so it is the + * register allocator that runs away when it cannot fit. */ + { + const char *e = getenv("SGX_MAX_TEMPS"); + + if (e && *e) { + unsigned m = (unsigned)strtoul(e, NULL, 0); + + if (m && m < opts.max_temps) + opts.max_temps = m; + } + } + /* And the texel is delivered the same way: this frame's texture unit + * samples per fragment and hands the result over as a primary + * attribute, so a sample is a read rather than an SMP. */ + /* The texture unit samples a whole iterated coordinate set and hands + * the texel over as a primary attribute. That only works when the + * coordinate *is* an iterated set: the vendor's own rule, recovered + * from its compiler, is a direct input read with an identity swizzle, + * no modifier and a plain TEX. Anything else - a computed coordinate, + * or one Mesa packed into a corner of another varying - has to be + * sampled by the program itself. + * + * Refusing it outright is what this did, and it is why every glmark2 + * scene that samples was refused: Mesa packs two varyings into one + * input, so the coordinate reads IN[0].yzxx and the swizzle alone + * disqualifies it. */ + /* Pre-iterated wherever the coordinate is iterated, which is the fast + * path - glmark2's texture scene runs at 56 frames a second on it and + * 2 on the shader-issued one. The shader-issued sample is only for a + * program the fast path cannot express, which was refused outright + * before: glmark2's texture scenes all failed for it. The heap + * corruption that forced this off has not reappeared since the frame's + * draw records and fixed addresses were fixed. SGX_NO_SMP puts the + * refusal back. */ + { + const char *e = getenv("SGX_TEX_PA"); + + if (e && *e) + opts.frag_tex_pa_base = atoi(e); + } + /* The shader as the backend receives it, written where usse-cg can be + * pointed at it. A refusal here is a compiler problem, and reproducing + * one from a running X server is otherwise a whole-machine exercise; + * the text round-trips through uir_parse, so the offline run compiles + * exactly what failed. */ + if (getenv("SGX_DUMP_UIR")) { + static unsigned seq; + const char *dir = getenv("SGX_DUMP_UIR"); + char path[512]; + int len = uir_print(s, NULL, 0); + char *text = len > 0 ? malloc((size_t)len + 1) : NULL; + FILE *f; + + if (text && uir_print(s, text, (size_t)len + 1) > 0) { + snprintf(path, sizeof path, "%s/%s-%u.uir", dir, + stage == UIR_STAGE_FRAGMENT ? "fs" : "vs", + seq++); + f = fopen(path, "w"); + if (f) { + fputs(text, f); + fclose(f); + fprintf(sgx_log(), "sgx: shader dumped to %s\n", + path); + } else { + fprintf(sgx_log(), "sgx: cannot write %s: %s\n", + path, strerror(errno)); + } + } + free(text); + } + /* Grown rather than fixed: a program that did not fit the buffer was + * refused, and how long a shader may be is not something the caller + * can be expected to know. Each attempt is a fresh codegen because the + * backend writes as it goes and cannot be resumed. */ + memset(&res, 0, sizeof res); + for (;;) { + unsigned cap = sh->code_cap ? sh->code_cap * 2u : + SGX_SHADER_CODE_INITIAL; + uint64_t *code; + + if (cap > SGX_SHADER_MAX_CODE) + break; + code = realloc(sh->code, (size_t)cap * sizeof *sh->code); + if (!code) { + uir_shader_free(s); + uir_codegen_result_fini(&res); + return SGX_SHADER_TOO_BIG; + } + sh->code = code; + sh->code_cap = cap; + + uir_codegen_result_fini(&res); + memset(&res, 0, sizeof(res)); + cgerr[0] = 0; + /* uir_codegen() returns the instruction count; negative is + * failure. */ + rc = uir_codegen(s, &opts, sh->code, cap, &res, cgerr, + sizeof cgerr); + /* Filling the buffer exactly is indistinguishable from running + * out of it, so that is grown into as well. */ + if (rc >= 0 && res.ninsns < cap) + break; + } + uir_shader_free(s); + free(tex_class); + free(tex_chunks); + if (rc < 0) { + if (err) + snprintf(err, errn, "%s", cgerr); + uir_codegen_result_fini(&res); + return SGX_SHADER_CODEGEN_FAILED; + } + if (getenv("SGX_DEBUG")) { + unsigned k; + + for (k = 0; k < res.ninsns && k < 512; k++) { + struct usse_insn in; + char text[256]; + + in.w0 = (uint32_t)(sh->code[k] & 0xffffffffu); + in.w1 = (uint32_t)(sh->code[k] >> 32); + text[0] = 0; + usse_disasm(&in, text, sizeof text); + fprintf(sgx_log(), "sgx: %2u %016llx %s\n", k, + (unsigned long long)sh->code[k], text); + } + } + if (res.ninsns > sh->code_cap) { + uir_codegen_result_fini(&res); + return SGX_SHADER_TOO_BIG; + } + + sh->ninsns = res.ninsns; + sh->ntemps = res.ntemps; + sh->src1_reg = res.src1_reg; + sh->has_src1 = res.has_src1 ? 1 : 0; + sh->nprimattr = res.nprimattr; + /* For the pixel data-master word: a task holding more temporaries + * holds fewer pixels. Set here rather than beside the other attribs + * fields above, because the count is not known until codegen has + * run - reading it earlier is reading an uninitialised result. */ + sh->attribs.ntemps = res.ntemps; + /* Coordinate sets stay on the coordinate iterators. The part has ten + * of them - EURASIA_PDS_DOUTI_USEISSUE_TC0..TC9, sgxdefs.h:3709-3718 - + * while V0 and V1 are the two *colour* iterators (:3719-3720), and a + * general varying carried on one does not arrive whole: glmark2's + * refract read MapCoord.x and vertex_position.z as exactly zero and + * drew half its bunny in ImageMap's clamped edge texel. + * + * The leading sets used to migrate onto V0 to satisfy "at most one + * coordinate set precedes the last", the rule recorded in + * work/iterated-component-limit. That rule was the sixteen-dword + * record truncation of the vertex copy program, fixed since, and with + * it gone three coordinate sets draw: refract renders 110601 colours + * where it had 23910, and jellyfish - the scene the migration was + * kept for - gains detail at 124461 against 112082. */ + sh->attribs.uses_kill = res.uses_kill; + sh->nsecattr = res.nsecattr; + + /* Flatten the pool into the dword order the driver uploads. The + * secondary PDS program DMAs this block into the sa bank; see + * work/uniform-abi/. */ + /* Truncating here would hand the hardware a program whose constants + * are partly whatever was in the bank, which renders wrongly rather + * than failing - so a pool that does not fit is an error. */ + if (res.npool > UIR_MAX_POOL) { + uir_codegen_result_fini(&res); + return SGX_SHADER_TOO_BIG; + } + for (i = 0; i < res.npool; i++) { + memcpy(&sh->pool[d], res.pool[i].bits, 4 * sizeof(uint32_t)); + d += 4; + } + sh->pool_dwords = d; + sh->pool_base = res.pool_base; + sh->smp_base = res.smp_base; + free(sh->smp_slot); + sh->smp_slot = NULL; + sh->nsamp = res.nsamp; + sh->nsmp_slots = res.nsmp_slots; + if (res.nsamp && res.smp_slot) { + sh->smp_slot = malloc(res.nsamp); + if (!sh->smp_slot) { + uir_codegen_result_fini(&res); + return SGX_SHADER_TOO_BIG; + } + memcpy(sh->smp_slot, res.smp_slot, res.nsamp); + } + sh->nuniform = res.nuniform; + free(sh->uni_base); + free(sh->uni_width); + sh->uni_base = NULL; + sh->uni_width = NULL; + if (res.nuniform) { + sh->uni_base = malloc(res.nuniform); + sh->uni_width = malloc(res.nuniform); + if (!sh->uni_base || !sh->uni_width) { + uir_codegen_result_fini(&res); + return SGX_SHADER_TOO_BIG; + } + memcpy(sh->uni_base, res.uni_base, res.nuniform); + memcpy(sh->uni_width, res.uni_width, res.nuniform); + } + sh->uses_kill = res.uses_kill; + sh->coord_f16 = res.coord_f16; + if (res.coord_f16) + fprintf(sgx_log(), "sgx: a sample packs its coordinate to f16 " + "in a temporary, which is SGX_SMP_F16_COORD's doing - " + "the unit ignores that operand and stays at texel " + "(0,0), so this program's textured draws come back one " + "flat colour. Unset it and the coordinate goes as f32 " + "registers, which the unit reads\n"); + uir_codegen_result_fini(&res); + sh->tex_preiterated = (unsigned)opts.frag_tex_preiterated; + + if (!ends(sh->code, sh->ninsns)) { + if (err) + snprintf(err, errn, + "last instruction lacks .end; it would run on"); + return SGX_SHADER_NO_END; + } + + if (res.nlimm > UIR_MAX_LIMM) + return SGX_SHADER_TOO_BIG; + memcpy(sh->limm, res.limm, res.nlimm * sizeof sh->limm[0]); + sh->nlimm = res.nlimm; + sh->compiled = 1; + return SGX_SHADER_OK; +} + +void sgx_shader_fini(struct sgx_shader *sh) +{ + if (!sh) + return; + free(sh->code); + free(sh->uni_base); + free(sh->uni_width); + free(sh->smp_slot); + sh->code = NULL; + sh->uni_base = NULL; + sh->uni_width = NULL; + sh->smp_slot = NULL; + sh->nsamp = sh->nsmp_slots = 0; + sh->nuniform = 0; + sh->code_cap = 0; + sh->ninsns = 0; + sh->compiled = 0; +} diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_shader.h mesa-26.2.2/src/gallium/drivers/sgx/sgx_shader.h --- mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_shader.h 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/sgx_shader.h 2026-09-08 10:57:36.681210836 +0200 @@ -0,0 +1,372 @@ +/* Shader objects: what a pipe_context binds, and what the driver has to do + * with the result. + * + * Compiling is only half of it. A compiled USSE program does not carry its + * constants inline - it reads them out of the secondary attribute bank, and + * the driver has to have put them there before the program runs. So a shader + * object is the code *and* the upload the driver owes it, and this is where + * the compiler's output stops being a blob and becomes something a frame can + * be built around. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#ifndef _SGX_SHADER_H_ +#define _SGX_SHADER_H_ + +/* The literal pool, in vec4 - what a fragment program's constants have to + * fit in, and what the screen advertises as its constant buffer size. */ +#define SGX_SHADER_MAX_POOL_VEC4 UIR_MAX_POOL + +/* How many registers the allocator may use: the vendor's own budget, which is + * also the most a task's count can be told about (SGX_TEMP_FIELD_MAX). Above + * it a program allocated registers the count could not name and the encoding + * could not hold. */ +#define SGX_SHADER_MAX_TEMPS 96 + +#include +#include + +#include "tgsi_to_uir.h" +#include "xpsb_frame.h" +#include "tgsi_parse.h" + +/* Where the code buffer starts and how far it may grow. It grows because a + * program that did not fit was refused outright, and real shaders - lighting, + * bump mapping - go past a few hundred instructions. The ceiling bounds a + * runaway compile rather than describing the hardware. */ +#define SGX_SHADER_CODE_INITIAL 512 +#define SGX_SHADER_MAX_CODE 8192 + +/* The record's two fixed quads, ahead of the coordinate sets: the vertex + * program writes position at 0 and colour at 4, and set k begins at 8 plus the + * widths of the sets before it. */ +#define SGX_VTX_POS_DW 0u +#define SGX_VTX_COLOUR_DW 4u +#define SGX_VTX_SET_DW 8u + +/* Experiment: where the coordinate sets begin in the record. The prefix is a + * position quad and a colour quad; SGX_SET_BASE moves that boundary so the + * offset a set starts at can be varied apart from its width. */ +static inline unsigned sgx_vtx_set_dw(void) +{ + const char *e = getenv("SGX_SET_BASE"); + + return e && *e ? (unsigned)atoi(e) : SGX_VTX_SET_DW; +} + +/* Gallium's semantic for a colour input, which is the one the layout has to + * tell apart from a coordinate set. */ +#define SGX_SEM_COLOR 1 + +struct sgx_shader { + int stage; /* UIR_STAGE_VERTEX or _FRAGMENT */ + /* The program as it was handed over, kept so it can be built again + * when state the compiler bakes in changes - an sRGB view on a unit + * makes the sample carry a conversion. Owned here, and outlives the + * compiled form; sgx_shader_from_tgsi_io() carries it across. */ + struct tgsi_insn *src_insns; + unsigned src_n; + struct tgsi_io *src_io; + unsigned srgb_key; /* the mask this form was built for */ + /* And which units carry a depth view, whose texel is three bytes of + * a 24-bit integer the program puts back together. */ + unsigned depth_key; + unsigned npot_key; /* and the wraps it does itself */ + /* The per-unit texture swizzles this form was built for, one entry + * per unit, grown to whatever the context binds. Kept beside the + * source for the same reason: a change of swizzle rebuilds. */ + uint16_t *swz_key; + unsigned nswz_key; + /* And what each unit's fetch delivers - the texel class and the + * plane count, SGX_TEXKEY() - which decides how the sample is + * unpacked and how many state blocks the unit takes. */ + unsigned char *cls_key; + unsigned ncls_key; + /* Per shadow sampler, the comparison the program was lowered with: + * 0 for none, else the pipe function plus one, the view swizzle + * above it (SGX_CMPKEY). The lowering is in NIR, ahead of the TGSI + * kept here, so a change rebuilds from the NIR the pipe keeps. */ + uint32_t *cmp_key; + unsigned ncmp_key; + /* The fragment program's NIR when it has a shadow sampler, owned + * here so the comparison can be lowered again for another sampler + * state; NULL otherwise. */ + void *src_nir; + /* The sampler units that NIR read through a shadow sampler. Kept + * apart from sampler_mask because the lowering can leave no sample + * at all - a comparison of ALWAYS is a constant, and the program + * that came out had no sampler for the rebuild to key on. */ + unsigned shadow_mask; + /* Which sampler units the program actually reads. A swizzle on a unit + * it never samples changes nothing about it, and comparing those too + * rebuilt every program on every draw. */ + unsigned sampler_mask; + uint64_t *code; /* code_cap entries, grown on demand */ + unsigned code_cap; + unsigned ninsns; + /* How many outputs a vertex program writes. The MTE emits four dwords + * per output, and group 10 has to name that count or the fragment + * task waits for dwords the vertex stage never sends. */ + unsigned nvtxout; + + /* What the driver must provide before the program runs. */ + unsigned ntemps; /* r registers it uses */ + unsigned nprimattr; /* pa registers it reads */ + /* A dual source blend's second colour: the register the program left + * it in, and whether it declared one at all. It is never written to + * the pixel - the blend reads it as SOP3's src0 - so it costs a quad + * of temporaries and nothing else. */ + unsigned src1_reg; + int has_src1; + unsigned nsecattr; /* sa registers it expects filled */ + unsigned nuniform; /* uniforms the program reads */ + /* Where each uniform sits in the bank and how wide it is. Mesa lays + * constants out a vec4 apiece; the bank packs them to the components + * actually read, so uploading means gathering into this. Both hold + * nuniform entries and are released by sgx_shader_fini(). */ + unsigned char *uni_base; + unsigned char *uni_width; + /* Ends by selecting against o0's incoming value (it discards), so the + * draw needs the read-modify-write object type. */ + unsigned uses_kill; + /* A sample in this program packs its coordinate to f16 in a + * temporary, which the unit ignores - every pixel reads texel + * (0,0). Reported so a flat fill is a line in the log rather than + * a picture to be puzzled over. */ + unsigned coord_f16; + /* What the frame must iterate for this program: which coordinate sets + * the record carries, how wide they are, and which are sampled. */ + struct xpsb_attribs attribs; + int fragcoord_set; /* which set carries gl_FragCoord, or -1 */ + /* The input carrying gl_FrontFacing, or -1. It takes no set: the + * program reads the ISP's backface bit itself, and which winding is + * front is baked in as face_key (sgx_shader_face_swap()), so a + * change of glFrontFace rebuilds it the way an sRGB view does. */ + int face_in; + unsigned face_key; + /* How many texture units sample each input, indexed by TGSI input. */ + unsigned char tex_units[16]; + /* Each input's own semantic and semantic index, as TGSI names them. + * The record is filled from the vertex program's outputs, and which + * output feeds which input is a question about semantics: matching + * them by position instead assumes the two stages declare things in + * the same order, which nothing requires. */ + unsigned char in_sem[16]; + unsigned char in_sem_idx[16]; + unsigned nin; + /* The texture unit iterates the coordinate and hands over a texel; + * clear when the program samples for itself, which needs the unit's + * state in the secondary attribute bank instead. */ + unsigned tex_preiterated; + /* Mesa packs a scalar varying in with a texture coordinate, so one + * varying has to serve as both the unit's coordinate and a value the + * program reads. A sampled set is three floats projected - its third + * component is the divisor - so the two cannot share one set. The + * varying is given one four-float set that is both sampled and + * iterated: the record carries the coordinate, then the one component + * the program reads, then a divisor of one. coord_c names the three + * original components in that order, coord_split_set the set. -1 when + * the varying is a coordinate and nothing else. */ + /* The sets were built with the sampled one first, so the texture issue + * leads the primary program the way every captured frame has it. */ + int sets_swapped; + /* Which fragment input feeds each coordinate set. Sampled sets are + * numbered first, because the set index is the texissue the frame + * hands the texture unit and the captured two-unit frame samples sets + * 0 and 1 - a coordinate pushed to set 2 by an iterated colour ahead + * of it was never observed to deliver. */ + uint8_t set_varying[XPSB_NSET_MAX]; + /* The input handed over as the packed colour, -1 for none. */ + int packed_in; + /* The program is a pass-through of an interpolated colour and nothing + * else - frag_in_is_packed_colour(). Every fixed-function fragment + * shader is this, and paired with a transform that writes a colour it + * is the shape that stops the part; sgx_pipe_settle_hwtcl() keeps it + * off the hardware path. */ + int colour_passthrough; + /* Which two components of that varying carry the set's coordinate. + * Normally 0 and 1; a coordinate Mesa packed behind another sits at + * 2 and 3, and the record moves them to the front. */ + uint8_t set_c0[XPSB_NSET_MAX], set_c1[XPSB_NSET_MAX]; + int coord_split_in; + int coord_split_set; + unsigned char coord_c[3]; + + /* The literal pool, in the dword order the driver uploads. */ + uint32_t pool[UIR_MAX_POOL * 4]; + unsigned pool_dwords; + unsigned pool_base; /* first sa register of the pool */ + /* Where the compiled code reads its sampler state from. Reported by + * the backend rather than worked out here: nuniform under-reports, so + * a descriptor placed from it landed on a uniform the shader samples + * with while the sample read zeros from the block the code names. */ + unsigned smp_base; + /* Each sampler's first state block and the total the code reads, + * from the backend: a unit stored in planes takes one a plane. */ + unsigned char *smp_slot; + unsigned nsamp, nsmp_slots; + + /* Which of the program's inputs are texture coordinates - a bitmask of + * TGSI input indices used as the coordinate of a sample. The frame's + * texture unit iterates the coordinate itself, out of the vertex + * record's coordinate slot, so the driver has to put the right varying + * there and this is how it knows which one. */ + unsigned tex_inputs; + + /* Uniforms materialised as immediates in the code. limm[k] says + * instruction limm[k].insn carries uniform dword limm[k].dword, and + * the driver patches each one as the values change - which is what + * lets a vertex program have more uniforms than an attribute bank + * holds. */ + struct uir_limm_ref limm[UIR_MAX_LIMM]; + unsigned nlimm; + + int compiled; +}; + +/* Releases what the shader allocated. Safe on a zeroed or failed shader, and + * required before the struct itself is freed. */ +void sgx_shader_fini(struct sgx_shader *sh); + +enum sgx_shader_status { + SGX_SHADER_OK = 0, + SGX_SHADER_TRANSLATE_FAILED, + SGX_SHADER_CODEGEN_FAILED, + SGX_SHADER_TOO_BIG, + SGX_SHADER_NO_END /* would run on past its last instruction */ +}; + +/* Compile a decoded TGSI program. err, when given, receives a message. */ +enum sgx_shader_status +sgx_shader_from_tgsi(struct sgx_shader *sh, int stage, + const struct tgsi_insn *insns, unsigned n, + const struct uir_binding *bindings, unsigned nbindings, + char *err, unsigned errn); + +/* The same, told what the shader's declarations said. A vertex program needs + * it: the record the fragment stage iterates has fixed slots, and which + * output goes in which is a question about semantics, not about numbering. */ +void sgx_shader_prefer_preiterated(int on); +/* Held across a set-compile-clear sequence: the sRGB mask and the iterator + * preference above are compile-wide, and Mesa compiles from several threads. */ +void sgx_shader_compile_lock(void); +void sgx_shader_compile_unlock(void); + +/* Which sampler units carry an sRGB view. Set before compiling a fragment + * program; the conversion GL requires on an sRGB sample is emitted into it, + * because the part has no sRGB texture format. */ +void sgx_shader_srgb_units(unsigned mask); +int sgx_shader_retarget_srgb(struct sgx_shader *sh, unsigned mask); +/* Rebuild a program that reads gl_FrontFacing for the other winding. */ +int sgx_shader_retarget_face(struct sgx_shader *sh, unsigned swap); +unsigned sgx_shader_srgb_units_get(void); + +/* Which sampler units carry a depth view. The texture unit has no depth + * format on this core and the ISP stores a 24-bit integer, so the sample + * comes back as four bytes and the program reassembles the depth. */ +void sgx_shader_depth_units(unsigned mask); +int sgx_shader_retarget_depth(struct sgx_shader *sh, unsigned mask); +unsigned sgx_shader_depth_units_get(void); + +/* Which units sample a texture whose width or height is not a power of two + * with a repeating wrap, by axis, so that the program wraps the coordinate + * itself - the unit's own REPEAT wraps at the padded power of two. One byte + * per treatment, indexed by texture unit. */ +#define SGX_NPOT_REPEAT_S(m) ((m) & 0xffu) +#define SGX_NPOT_REPEAT_T(m) (((m) >> 8) & 0xffu) +#define SGX_NPOT_MIRROR_S(m) (((m) >> 16) & 0xffu) +#define SGX_NPOT_MIRROR_T(m) (((m) >> 24) & 0xffu) +#define SGX_NPOT_KEY(rs, rt, ms, mt) \ + (((rs) & 0xffu) | (((rt) & 0xffu) << 8) | \ + (((ms) & 0xffu) << 16) | (((mt) & 0xffu) << 24)) + +void sgx_shader_npot_units(unsigned mask); +unsigned sgx_shader_npot_units_get(void); +int sgx_shader_retarget_npot(struct sgx_shader *sh, unsigned mask); +/* Whether a program compiled now reads the face bit inverted: set from the + * rasterizer's winding and the target's flip, sgx_face_swap_of(). */ +void sgx_shader_face_swap(int swap); + +/* What a sampled value means is the view's business, not the texture's: + * GL_ALPHA reads (0,0,0,a) out of the one byte the part fetched and + * GL_LUMINANCE reads (l,l,l,1). This core has no texture swizzle - + * SGX_FEATURE_TAG_SWIZZLE is 543 and later - so it belongs in the program, + * which makes the program depend on the views bound under it. + * + * One entry per unit: four channels, three bits each, 0-3 selecting a + * sampled channel and 4 and 5 the constants zero and one. That is Gallium's + * own PIPE_SWIZZLE numbering, mapped explicitly rather than assumed. */ +#define SGX_SWZ_ZERO 4u +#define SGX_SWZ_ONE 5u +#define SGX_SWZ_CHAN(k, i) (((k) >> ((i) * 3)) & 7u) +#define SGX_SWZ_MAKE(r, g, b, a) \ + ((uint16_t)((r) | ((g) << 3) | ((b) << 6) | ((a) << 9))) +#define SGX_SWZ_IDENTITY SGX_SWZ_MAKE(0u, 1u, 2u, 3u) + +/* Registers one sampler state block takes in the secondary bank: + * {ctl, fmt, addr}. The backend's SMP_SLOT, which sgx_context.h asserts + * its own copy against. */ +#define SGX_SMP_SLOT_REGS 3u + +/* Where a sample of unit `unit`, plane `chunk`, reads its {ctl, fmt, addr} + * block: the layout the backend laid down, so the descriptor writer and the + * code cannot disagree about it. nu is how many units the caller means to + * describe, which only the fallback for a program that reported no layout + * needs. *nchunks, when given, receives the unit's plane count. + * + * The two numbers this returns and the sa register the compiled SMP names + * are checked against each other by the shader host test: they were four + * registers apart on the float textures, and the part read the shifted + * words as an address and faulted. */ +unsigned sgx_shader_smp_reg(const struct sgx_shader *sh, unsigned unit, + unsigned chunk, unsigned nu, unsigned *nchunks); + +void sgx_shader_swz_units(const uint16_t *swz, unsigned n); +int sgx_shader_swz_differs(const struct sgx_shader *sh, const uint16_t *swz, + unsigned n); +int sgx_shader_retarget_swz(struct sgx_shader *sh, const uint16_t *swz, + unsigned n); + +/* What a unit's fetch returns, one byte per unit: the compiler's + * UIR_TEXCLASS_* in the low nibble and the plane count less one above it. + * Zero is a packed 8888 texel in one plane, the identity the lookup + * supplies past the end of an array. */ +#define SGX_TEXKEY(cls, chunks) ((unsigned char)((cls) | (((chunks) - 1u) << 4))) +#define SGX_TEXKEY_CLASS(k) ((k) & 0xfu) +#define SGX_TEXKEY_CHUNKS(k) ((((k) >> 4) & 0xfu) + 1u) +void sgx_shader_cls_units(const unsigned char *cls, unsigned n); +int sgx_shader_cls_differs(const struct sgx_shader *sh, + const unsigned char *cls, unsigned n); +int sgx_shader_retarget_cls(struct sgx_shader *sh, const unsigned char *cls, + unsigned n); + +/* A shadow sampler's key: the comparison plus one in the low byte, the + * view swizzle above it; zero for a unit with no comparison. */ +#define SGX_CMPKEY(func, swz) ((uint32_t)(((func) + 1u) | ((uint32_t)(swz) << 8))) +#define SGX_CMPKEY_FUNC(k) (((k) & 0xffu) - 1u) +#define SGX_CMPKEY_SWZ(k) ((uint16_t)((k) >> 8)) +int sgx_shader_cmp_differs(const struct sgx_shader *sh, const uint32_t *cmp, + unsigned n); +void sgx_shader_vtx_layout(const unsigned char *sem, const unsigned char *idx, + const unsigned char *slot, unsigned n); + +/* The same with the emitted dword each output starts at, so a coordinate set + * narrower than four floats does not leave the sets behind it where the + * iterator does not read them. */ +void sgx_shader_vtx_layout_dw(const unsigned char *sem, + const unsigned char *idx, + const unsigned char *slot, + const unsigned char *dw, unsigned n); +void sgx_shader_vtx_pack_colour(int on); + +enum sgx_shader_status +sgx_shader_from_tgsi_io(struct sgx_shader *sh, int stage, + const struct tgsi_insn *insns, unsigned n, + const struct uir_binding *bindings, unsigned nbindings, + const struct tgsi_io *io, char *err, unsigned errn); + +const char *sgx_shader_status_name(enum sgx_shader_status st); + +#endif diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_state.c mesa-26.2.2/src/gallium/drivers/sgx/sgx_state.c --- mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_state.c 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/sgx_state.c 2026-09-08 10:57:36.681224687 +0200 @@ -0,0 +1,1056 @@ +/* SPDX-License-Identifier: MIT */ +/* + * Copyright (C) 2026 René Rebe + */ +#include "sgx_state.h" +#include +#include + +/* GL-IMPLEMENTATION.md section 3, and the DDK's EURASIA_ISPA_* (sgxdefs.h): + * [24:22] depth compare function, GL order NEVER=0 .. ALWAYS=7 + * [20] depth WRITE DISABLE - note the inversion against Gallium's + * writemask, which is the kind of thing that renders almost right + * [11] 2SIDED: a back-face set follows the front one (:1375) + * [9] BPRES: word B follows (:1377) + * [8] CPRES: word C follows (:1378) + * [7:0] stencil reference, SREF (:1380) + */ +static uint32_t isp_depth_word(const struct sgx_dsa_state *dsa) +{ + uint32_t w = 0; + + /* A disabled depth test is ALWAYS with writes off, which is exactly + * what the captured frame holds: 0x01d00000. */ + w |= (uint32_t)(dsa->depth_enabled ? (dsa->depth_func & 7) + : SGX_FUNC_ALWAYS) << 22; + if (!dsa->depth_writemask || !dsa->depth_enabled) + w |= 1u << 20; + return w; +} + +/* One face's reference and, when its word C says anything the hardware's + * default does not, CPRES - the vendor's rule (validate.c:3887-3890). */ +static uint32_t isp_face_word(const struct sgx_dsa_state *dsa, unsigned face, + uint32_t depth) +{ + uint32_t w = depth; + + if (!sgx_isp_has_stencil(dsa)) + return w; + w |= (face ? dsa->stencil_ref_back : dsa->stencil_ref) & 0xff; + if (sgx_isp_word2(&dsa->stencil[face]) != SGX_ISPC_DEFAULT) + w |= SGX_ISP_CPRES; + return w; +} + +uint32_t sgx_isp_word0_face(const struct sgx_dsa_state *dsa, unsigned face) +{ + return isp_face_word(dsa, face, isp_depth_word(dsa)); +} + +uint32_t sgx_isp_word0(const struct sgx_dsa_state *dsa) +{ + return sgx_isp_word0_face(dsa, 0); +} + +unsigned sgx_isp_stencil_face(const struct sgx_dsa_state *dsa, + const struct sgx_rasterizer_state *r) +{ + if (!dsa->stencil[1].enabled) + return 0; + /* The front faces are culled, so only the back set can matter. */ + return (r && r->cull_face == 1) ? 1 : 0; +} + +/* The MTE culls by device-space winding and the vendor's GL_BACK under a CCW + * front face is CULLMODE_CW (validate.c:3541-3545), so the part's clockwise + * face is GL's back face there. The ISP's two-sided front is that same + * clockwise face: the vendor hands the front set GL's front stencil only + * under glFrontFace(GL_CW) (:3739-3757), and inverts the winding again for a + * y-flipped surface (:3514-3524). */ +int sgx_isp_front_is_gl_front(const struct sgx_rasterizer_state *r) +{ + if (!r) + return 0; + return ((r->front_ccw ^ r->target_y_flipped) & 1u) == 0; +} + +/* The vendor's conversion (validate.c:3590-3601): each of glPolygonOffset's + * two floats, times an app hint that defaults to 1.0 (misc.c:627-631), + * truncated to an integer and clamped to the signed five-bit field. The + * units are the part's own depth steps, not GL's minimum resolvable + * difference. */ +static uint32_t isp_bias_field(float v, unsigned shift) +{ + int i = (int)v; + + if (i < -16) + i = -16; + if (i > 15) + i = 15; + return ((uint32_t)i & 0x1fu) << shift; +} + +uint32_t sgx_isp_word1(const struct sgx_rasterizer_state *r, int hw_transform, + unsigned colormask) +{ + uint32_t w = 0; + unsigned key; + + if (r && r->offset_tri && hw_transform) { + w |= isp_bias_field(r->offset_units, SGX_ISPB_DBIASUNITS_SHIFT); + w |= isp_bias_field(r->offset_scale, SGX_ISPB_DBIASFACTOR_SHIFT); + } + /* UPASSCTL is the vendor's GLES2_COLORMASK set (state.h:103-106, the + * branch without VEC34): red at the top, alpha at the bottom - the + * reverse of Mesa's PIPE_MASK_* nibble. */ + key = ((colormask & 0x1u) ? 8u : 0u) | ((colormask & 0x2u) ? 4u : 0u) | + ((colormask & 0x4u) ? 2u : 0u) | ((colormask & 0x8u) ? 1u : 0u); + w |= key << SGX_ISPB_UPASSCTL_SHIFT; + return w; +} + +/* validate.c:3722-3760. One set when the two faces agree, or when a cull + * leaves only one of them to matter; both, under 2SIDED, when they differ + * and nothing is culled. The back set carries every word (:2664-2677), so + * the front set is given B as well rather than leave a B behind an absent + * one. */ +void sgx_isp_sets(const struct sgx_dsa_state *dsa, + const struct sgx_rasterizer_state *r, int hw_transform, + unsigned colormask, struct sgx_isp_set *ff, + struct sgx_isp_set *bf) +{ + uint32_t depth = isp_depth_word(dsa); + unsigned face = 0, gl_front; + int two; + + ff->a = depth; + ff->b = sgx_isp_word1(r, hw_transform, colormask); + ff->c = SGX_ISPC_DEFAULT; + bf->a = 0; + bf->b = ff->b; + bf->c = SGX_ISPC_DEFAULT; + if (ff->b != SGX_ISPB_DEFAULT) + ff->a |= SGX_ISP_BPRES; + if (!sgx_isp_has_stencil(dsa)) + return; + two = dsa->stencil[1].enabled && + (sgx_isp_word2(&dsa->stencil[1]) != + sgx_isp_word2(&dsa->stencil[0]) || + dsa->stencil_ref_back != dsa->stencil_ref); + if (two && r && r->cull_face) { + face = sgx_isp_stencil_face(dsa, r); + two = 0; + } + if (!two) { + ff->a = isp_face_word(dsa, face, ff->a); + ff->c = sgx_isp_word2(&dsa->stencil[face]); + return; + } + gl_front = sgx_isp_front_is_gl_front(r) ? 0u : 1u; + ff->a |= SGX_ISP_2SIDED | SGX_ISP_BPRES; + bf->a = ff->a; + ff->a = isp_face_word(dsa, gl_front, ff->a); + ff->c = sgx_isp_word2(&dsa->stencil[gl_front]); + bf->a = isp_face_word(dsa, gl_front ^ 1u, bf->a); + bf->c = sgx_isp_word2(&dsa->stencil[gl_front ^ 1u]); +} + +int sgx_isp_set_vistest(struct sgx_isp_set *ff, struct sgx_isp_set *bf, int reg) +{ + if (reg < 0) + return 0; + if (!ff || (unsigned)reg >= SGX_ISP_VISTEST_REGS) + return -EINVAL; + ff->b = (ff->b & ~SGX_ISPB_VISREG_MASK) | SGX_ISPB_VISTEST | + (unsigned)reg; + ff->a |= SGX_ISP_BPRES; + if (bf && (ff->a & SGX_ISP_2SIDED)) { + bf->b = (bf->b & ~SGX_ISPB_VISREG_MASK) | SGX_ISPB_VISTEST | + (unsigned)reg; + bf->a |= SGX_ISP_BPRES; + } + return 0; +} + +/* The object type and, for a point or a line, its width. Both live in ISP + * word A: the type at [18:15] and the width at [31:28] as width - 1. + * + * Note this is a different field from the one sgx_context.c calls the object + * type when it stamps a record read-modify-write - that is the pass type at + * [27:25], which says how the ISP schedules the object rather than what shape + * it is. */ +uint32_t sgx_isp_objtype(uint32_t word0, unsigned objtype, unsigned width) +{ + word0 = (word0 & ~SGX_ISP_OBJTYPE_MASK) | + ((objtype << SGX_ISP_OBJTYPE_SHIFT) & SGX_ISP_OBJTYPE_MASK); + if (objtype != SGX_ISP_OBJ_TRI) { + if (!width) + width = 1u; + if (width > SGX_ISP_PLWIDTH_MAX) + width = SGX_ISP_PLWIDTH_MAX; + word0 = (word0 & ~SGX_ISP_PLWIDTH_MASK) | + (((width - 1u) << SGX_ISP_PLWIDTH_SHIFT) & + SGX_ISP_PLWIDTH_MASK); + } + return word0; +} + +int sgx_isp_has_stencil(const struct sgx_dsa_state *dsa) +{ + return dsa->stencil[0].enabled || dsa->stencil[1].enabled; +} + +/* The hardware's stencil ops are ordered + * 0 KEEP 1 ZERO 2 REPLACE 3 INCR 4 DECR 5 INVERT 6 INCR_WRAP 7 DECR_WRAP + * and Gallium's are + * 0 KEEP 1 ZERO 2 REPLACE 3 INCR 4 DECR 5 INCR_WRAP 6 DECR_WRAP 7 INVERT + * so the last three differ. A wrong mapping here still renders - it just + * renders the wrong stencil - which is why it is a table rather than a cast. + */ +static uint32_t hw_op(unsigned gallium_op) +{ + static const uint8_t map[8] = { 0, 1, 2, 3, 4, 6, 7, 5 }; + + return map[gallium_op & 7]; +} + +/* [27:25] stencil compare func, [24:22] sfail, [21:19] zfail, [18:16] zpass, + * [15:8] compare mask, [7:0] write mask. */ +uint32_t sgx_isp_word2(const struct sgx_stencil_face *f) +{ + return ((uint32_t)(f->func & 7) << 25) | + (hw_op(f->fail_op) << 22) | + (hw_op(f->zfail_op) << 19) | + (hw_op(f->zpass_op) << 16) | + ((uint32_t)(f->valuemask & 0xff) << 8) | + (uint32_t)(f->writemask & 0xff); +} + +/* [1:0] cull: 0 none, otherwise 1 + (winding XOR face XOR flipped) - the + * hardware discards by device-space signed area, so which value culls which + * face depends on the winding and on whether the target is y-flipped. Getting + * this from three booleans rather than a lookup is why it is written out. + * [18:17] shade model, the DDK's EURASIA_MTE_SHADE_* on this core: 0 GOURAUD, + * then VERTEX0, VERTEX1, VERTEX2 - flat, taking the colour from the corner + * named. GL's provoking vertex is the last corner, or the first under + * glProvokingVertex(GL_FIRST_VERTEX_CONVENTION); the decomposition of strips, + * fans and quads puts it at the corner this field names. */ +uint32_t sgx_raster_word(const struct sgx_rasterizer_state *r) +{ + uint32_t w = 0; + + if (r->cull_face) { + unsigned back = r->cull_face == 2; + unsigned v = 1u + ((r->front_ccw ^ back ^ r->target_y_flipped) & 1u); + + w |= v & 3u; + } + if (r->flatshade) + w |= (r->flatshade_first ? 1u : 3u) << SGX_MTE_SHADE_SHIFT; + /* Bit 16 says the vertices arrived already transformed, which on this + * driver's ordinary path they did: the draw module transforms on the + * CPU and the frame's vertex program is a pass-through. The only + * capture of an enabled cull in the tree carries 0x00010001 - cull + * value 1 with this bit set - and emitting the group without it + * discards every triangle whichever winding was selected. + * + * It describes the vertices, not the culling, so it belongs to the + * group rather than to the cull field: a flat-shaded draw that culls + * nothing still emits the group for the shade model, and setting the + * bit only alongside a cull face left every such draw with no + * geometry at all - glxheads draws one flat-shaded triangle and + * nothing but its clear ever reached the screen. A word of zero still + * means the group is not emitted. */ + if (w && !r->hw_transform) + w |= 1u << 16; + return w; +} + +/* 1 + the corner a flat value comes from, or 0 when the draw is gouraud. The + * MTE names it here, the iterator's DOUTI FLATSHADE field and the vertex data + * master's index list word each name it in their own encoding, and the vendor + * writes all three from the one shade model (opengles1/usegles.c:2240-2252, + * validate.c:4476-4482, validate.c:5013-5025). */ +unsigned sgx_raster_flat_corner(uint32_t raster_word) +{ + return (raster_word & SGX_MTE_SHADE_MASK) >> SGX_MTE_SHADE_SHIFT; +} + +/* Blending, from the factors rather than from a table of named operators. + * + * The equation lives in the fragment program: the vendor appends one of + * three integer sum-of-products instructions to it, and which one depends on + * what the factors need (GL-IMPLEMENTATION.md "Blending", opengles2/use.c + * CreateFBBlendUSECode). Every field below is the DDK's: + * + * SOP2 (op 0x10, sgxdefs.h:5648-5745) two operands, separate colour and + * alpha selects: hi [8:6] CSEL1, [24] CMOD1, [5:3] CSEL2, [15] CMOD2, + * [21:20] ASEL1, [11] AMOD1, [10:9] ASEL2, [2] AMOD2; lo [19:18] COP, + * [17:16] AOP (0 ADD 1 SUB 2 MIN 3 MAX), [14] ADSTMOD - negate the + * alpha result, which is how SUB and REVERSE_SUBTRACT mix. + * CSEL: 0 zero 1 src1 2 src2 3 src1.a 4 src2.a 5 min(src1.a,1-src2.a). + * ASEL: 0 zero 1 src1.a 2 src2.a. + * SOPWM (op 0x12, :5751-5822) one select pair for all four channels and a + * write mask hi [14:11] (R 8, G 4, B 2, A 1); hi [21:20] COP, [10:9] + * AOP, [8:6] SEL1, [5:3] SEL2, [24]/[15] the complements. + * SEL: 0 zero 1 asat 2 src1 3 src1.a 6 src2 7 src2.a. + * SOP3 (op 0x11, :5828-5909) a third operand, src0 in lo [20:14] with its + * bank in hi [2]: CSEL adds 4 src0 and 5 src0.a; alpha is hi [13:12] + * ASEL1 (0 zero 1 src0.a 2 src1.a 3 src2.a) with [14] AMOD1, and the + * alpha destination factor is CSEL2's alpha lane. COP has ADD and SUB + * only; hi [11] (DESTMOD) selects the ARSOP co-issue, which BRN + * 25060 breaks on this core with a complemented CSEL2, so it stays + * clear. The captured single-instruction constant-colour path is + * "mov rC, sa1 ; sop3 o0, rC, rN, o0" (work/blendfb/). + * + * REVERSE_SUBTRACT is SUB with the operand banks swapped (lo [31:30] and + * [29:28]) and every select re-expressed against the swapped registers - the + * vendor's EncodeTwoSourceBlend() and its aui32SOP2FlipSel. GL_ONE is "zero, + * complemented" in every form. + * + * The register names are patched here, not by the caller: the source colour + * is `src` (a temporary), the destination is o0 - readable as a source under + * the read-modify-write object type - and the constant, when a factor needs + * it, is loaded into a temporary with LIMM first (sgxdefs.h:7437-7447), the + * way the vendor stages it from the secondary attribute bank. + */ +#define SGX_USE1_SKIPINV (1u << 23) +#define SGX_USE1_END (1u << 18) +#define SGX_USE1_OP(op) ((uint32_t)(op) << 27) +#define SGX_USE0_DST(r) (((uint32_t)(r) & 0x7fu) << 21) +#define SGX_USE0_SRC0(r) (((uint32_t)(r) & 0x7fu) << 14) +#define SGX_USE0_SRC1(r) (((uint32_t)(r) & 0x7fu) << 7) +#define SGX_USE0_SRC2(r) ((uint32_t)(r) & 0x7fu) +#define SGX_USE0_S1BANK(b) ((uint32_t)(b) << 30) +#define SGX_USE0_S2BANK(b) ((uint32_t)(b) << 28) +#define SGX_BANK_TEMP 0u +#define SGX_BANK_OUTPUT 1u + +/* The captured SOP2 - over, premultiplied - and the swap it keys on. */ +#define SGX_BLEND_LO_NORMAL 0x10000000u +#define SGX_BLEND_LO_REVSUB 0x40000000u + +/* A blend factor, taken apart: what it reads, which channel, complemented. */ +enum sgx_bf_kind { SGX_BK_ZERO, SGX_BK_SRC, SGX_BK_DST, SGX_BK_CONST, + SGX_BK_ASAT, SGX_BK_SRC1 }; + +struct sgx_bf { + unsigned char kind, alpha, inv; +}; + +/* PIPE_BLENDFACTOR_*: bit 4 is the complement, [3:0] the operand; ONE is + * zero complemented. SRC1_* is the fragment program's second colour, which + * reaches the blend in SOP3's src0 - the same slot the blend constant uses, + * so a state that wants both is refused rather than mis-encoded. */ +static int bf_decode(unsigned f, struct sgx_bf *b) +{ + b->inv = (f & 0x10u) != 0; + b->alpha = 0; + switch (f & 0xfu) { + case 0x1: b->kind = SGX_BK_ZERO; b->inv = !b->inv; return 0; + case 0x2: b->kind = SGX_BK_SRC; return 0; + case 0x3: b->kind = SGX_BK_SRC; b->alpha = 1; return 0; + case 0x4: b->kind = SGX_BK_DST; b->alpha = 1; return 0; + case 0x5: b->kind = SGX_BK_DST; return 0; + case 0x6: b->kind = SGX_BK_ASAT; return b->inv ? -1 : 0; + case 0x7: b->kind = SGX_BK_CONST; return 0; + case 0x8: b->kind = SGX_BK_CONST; b->alpha = 1; return 0; + case 0x9: b->kind = SGX_BK_SRC1; return 0; + case 0xa: b->kind = SGX_BK_SRC1; b->alpha = 1; return 0; + default: return -1; + } +} + +/* The ISP's translucent pass type: any factor that reads the destination. + * A write mask or a logic op does not make an object translucent, and neither + * does a dual-source factor - the second colour is the program's own. */ +int sgx_blend_translucent(unsigned rgb_src, unsigned rgb_dst, + unsigned alpha_src, unsigned alpha_dst) +{ + struct sgx_bf s, a; + + if (rgb_dst != SGX_BF_ZERO || alpha_dst != SGX_BF_ZERO) + return 1; + /* A source factor that reads the destination, or its alpha. */ + return (!bf_decode(rgb_src, &s) && + (s.kind == SGX_BK_DST || s.kind == SGX_BK_ASAT)) || + (!bf_decode(alpha_src, &a) && + (a.kind == SGX_BK_DST || a.kind == SGX_BK_ASAT)); +} + +static const struct sgx_bf bf_one = { SGX_BK_ZERO, 0, 1 }; + +/* What the factor contributes to the alpha channel: a colour select's alpha + * lane is that operand's alpha, and saturate's is one (GL, and the vendor's + * ASEL table). */ +static struct sgx_bf bf_lane(struct sgx_bf b) +{ + if (b.kind == SGX_BK_ASAT) + return bf_one; + if (b.kind != SGX_BK_ZERO) + b.alpha = 1; + return b; +} + +static int bf_same(struct sgx_bf a, struct sgx_bf b) +{ + return a.kind == b.kind && a.alpha == b.alpha && a.inv == b.inv; +} + +static int bf_is_one(struct sgx_bf b) +{ + return b.kind == SGX_BK_ZERO && b.inv; +} + +/* A select code for a factor whose referent sits in operand `slot` (0, 1 or + * 2). SOP2's enum differs from the one SOPWM and SOP3 share. */ +static int sel_sop2(struct sgx_bf b, unsigned slot, unsigned *sel) +{ + switch (b.kind) { + case SGX_BK_ZERO: *sel = 0; return 0; + case SGX_BK_ASAT: *sel = 5; return slot == 1 ? 0 : -1; + case SGX_BK_SRC: + case SGX_BK_DST: *sel = (b.alpha ? 2u : 0u) + slot; + return slot == 1 || slot == 2 ? 0 : -1; + /* SGX_BK_CONST and SGX_BK_SRC1 both live in src0, which this + * instruction does not have. */ + default: return -1; + } +} + +static int asel_sop2(struct sgx_bf b, unsigned slot, unsigned *sel) +{ + b = bf_lane(b); + if (b.kind == SGX_BK_ZERO) { + *sel = 0; + return 0; + } + if ((b.kind != SGX_BK_SRC && b.kind != SGX_BK_DST) || + (slot != 1 && slot != 2)) + return -1; /* src0's factors, again */ + *sel = slot; + return 0; +} + +static int sel_sop3(struct sgx_bf b, unsigned slot, unsigned *sel) +{ + static const unsigned char base[3] = { 4, 2, 6 }; + + switch (b.kind) { + case SGX_BK_ZERO: *sel = 0; return 0; + case SGX_BK_ASAT: *sel = 1; return slot == 1 ? 0 : -1; + default: + if (slot > 2) + return -1; + *sel = base[slot] + (b.alpha ? 1u : 0u); + return 0; + } +} + +static int asel_sop3(struct sgx_bf b, unsigned slot, unsigned *sel) +{ + b = bf_lane(b); + if (b.kind == SGX_BK_ZERO) { + *sel = 0; + return 0; + } + if (slot > 2) + return -1; + *sel = slot + 1u; + return 0; +} + +/* One sum-of-products instruction, before encoding. Selects are in the + * form's own enum; sel/mod 1 multiply operand 1, sel/mod 2 operand 2. */ +struct sgx_sop { + unsigned dbank, dst; + unsigned s1bank, s1, s2bank, s2, s0; /* s0 is always a temporary */ + unsigned csel1, cmod1, csel2, cmod2; + unsigned asel1, amod1, asel2, amod2; + unsigned cop, aop, aneg; + unsigned mask; /* SOPWM: R 8 G 4 B 2 A 1 */ +}; + +static uint32_t sop_lo(const struct sgx_sop *s) +{ + return SGX_USE0_S1BANK(s->s1bank) | SGX_USE0_S2BANK(s->s2bank) | + SGX_USE0_DST(s->dst) | SGX_USE0_SRC1(s->s1) | + SGX_USE0_SRC2(s->s2); +} + +static uint64_t enc_sop2(const struct sgx_sop *s) +{ + uint32_t hi = SGX_USE1_OP(0x10) | SGX_USE1_SKIPINV | (s->dbank & 3u) | + ((s->csel1 & 7u) << 6) | (s->cmod1 ? 1u << 24 : 0u) | + ((s->csel2 & 7u) << 3) | (s->cmod2 ? 1u << 15 : 0u) | + ((s->asel1 & 3u) << 20) | (s->amod1 ? 1u << 11 : 0u) | + ((s->asel2 & 3u) << 9) | (s->amod2 ? 1u << 2 : 0u); + uint32_t lo = sop_lo(s) | ((s->cop & 3u) << 18) | + ((s->aop & 3u) << 16) | (s->aneg ? 1u << 14 : 0u); + + return ((uint64_t)hi << 32) | lo; +} + +static uint64_t enc_sopwm(const struct sgx_sop *s) +{ + uint32_t hi = SGX_USE1_OP(0x12) | SGX_USE1_SKIPINV | (s->dbank & 3u) | + ((s->cop & 3u) << 20) | ((s->aop & 3u) << 9) | + ((s->mask & 0xfu) << 11) | + ((s->csel1 & 7u) << 6) | (s->cmod1 ? 1u << 24 : 0u) | + ((s->csel2 & 7u) << 3) | (s->cmod2 ? 1u << 15 : 0u); + + return ((uint64_t)hi << 32) | sop_lo(s); +} + +static uint64_t enc_sop3(const struct sgx_sop *s) +{ + uint32_t hi = SGX_USE1_OP(0x11) | SGX_USE1_SKIPINV | (s->dbank & 3u) | + ((s->cop & 3u) << 20) | ((s->aop & 3u) << 9) | + ((s->csel1 & 7u) << 6) | (s->cmod1 ? 1u << 24 : 0u) | + ((s->csel2 & 7u) << 3) | (s->cmod2 ? 1u << 15 : 0u) | + ((s->asel1 & 3u) << 12) | (s->amod1 ? 1u << 14 : 0u); + + return ((uint64_t)hi << 32) | (sop_lo(s) | SGX_USE0_SRC0(s->s0)); +} + +/* limm rN, #imm: op 0x1f, OTHER (2) at [21:20], LIMM (4) at [26:24]; the + * immediate is lo [20:0], hi [8:4] and hi [17:12]. */ +static uint64_t enc_limm(unsigned reg, uint32_t imm) +{ + uint32_t hi = SGX_USE1_OP(0x1f) | (4u << 24) | (2u << 20) | + SGX_USE1_SKIPINV | (((imm >> 21) & 0x1fu) << 4) | + (((imm >> 26) & 0x3fu) << 12); + + return ((uint64_t)hi << 32) | SGX_USE0_DST(reg) | (imm & 0x1fffffu); +} + +/* PIPE_BLEND_*: the two-bit op and whether the operands are reversed. */ +static int blend_op(unsigned func, unsigned *op, unsigned *rev) +{ + *rev = 0; + switch (func) { + case 0: *op = 0; return 0; /* ADD */ + case 1: *op = 1; return 0; /* SUBTRACT */ + case 2: *op = 1; *rev = 1; return 0; /* REVERSE_SUBTRACT */ + case 3: *op = 2; return 0; /* MIN */ + case 4: *op = 3; return 0; /* MAX */ + default: return -1; + } +} + +/* Where a factor's referent is, given which operand holds the source. */ +static unsigned bf_slot(struct sgx_bf b, unsigned src_slot) +{ + switch (b.kind) { + case SGX_BK_SRC: + case SGX_BK_ASAT: return src_slot; + case SGX_BK_DST: return 3u - src_slot; + default: return 0; /* the constant, SOP3's src0 */ + } +} + +/* Mesa's mask (R 1, G 2, B 4, A 8) onto the SOPWM nibble, which runs the + * other way (EURASIA_USE1_SOP2WM_WRITEMASK_R is 8, _A is 1). */ +static unsigned wm_nibble(unsigned mask) +{ + return ((mask & 1u) ? 8u : 0u) | ((mask & 2u) ? 4u : 0u) | + ((mask & 4u) ? 2u : 0u) | ((mask & 8u) ? 1u : 0u); +} + +/* The list being built. */ +struct sgx_bl { + uint64_t *out; + unsigned n, cap; + int err; +}; + +static void bl_put(struct sgx_bl *l, uint64_t insn) +{ + if (l->n >= l->cap) { + l->err = -ENOSPC; + return; + } + l->out[l->n++] = insn; +} + +/* The equation itself: op1 * f1 op op2 * f2 into dst, as one SOP2, or as + * two SOPWMs (RGB, then A) when `split` asks for each channel group to carry + * its own equation and swap. A constant factor cannot reach here. */ +static int bl_equation(struct sgx_bl *l, struct sgx_bf fs, struct sgx_bf fd, + struct sgx_bf as, struct sgx_bf ad, + unsigned cop, unsigned crev, unsigned aop, unsigned arev, + unsigned op1bank, unsigned op1, unsigned op2bank, + unsigned op2, unsigned dbank, unsigned dst, int split) +{ + struct sgx_sop s; + unsigned swap, src_slot; + + memset(&s, 0, sizeof s); + s.dbank = dbank; + s.dst = dst; + if (!split) { + /* The vendor's rule: the swap follows the colour equation, and + * the alpha subtraction is negated when its direction differs + * from what the swap gave it. */ + swap = crev; + s.aneg = aop == 1u && arev != swap; + src_slot = swap ? 2u : 1u; + s.s1bank = swap ? op2bank : op1bank; + s.s1 = swap ? op2 : op1; + s.s2bank = swap ? op1bank : op2bank; + s.s2 = swap ? op1 : op2; + s.cop = cop; + s.aop = aop; + if (sel_sop2(swap ? fd : fs, bf_slot(swap ? fd : fs, src_slot), + &s.csel1) || + sel_sop2(swap ? fs : fd, bf_slot(swap ? fs : fd, src_slot), + &s.csel2) || + asel_sop2(swap ? ad : as, bf_slot(swap ? ad : as, src_slot), + &s.asel1) || + asel_sop2(swap ? as : ad, bf_slot(swap ? as : ad, src_slot), + &s.asel2)) + return -ENOTSUP; + s.cmod1 = (swap ? fd : fs).inv; + s.cmod2 = (swap ? fs : fd).inv; + s.amod1 = bf_lane(swap ? ad : as).inv; + s.amod2 = bf_lane(swap ? as : ad).inv; + bl_put(l, enc_sop2(&s)); + return l->err; + } + /* RGB, with its own swap. */ + swap = crev; + src_slot = swap ? 2u : 1u; + s.s1bank = swap ? op2bank : op1bank; + s.s1 = swap ? op2 : op1; + s.s2bank = swap ? op1bank : op2bank; + s.s2 = swap ? op1 : op2; + s.cop = s.aop = cop; + s.mask = 0xeu; + if (sel_sop3(swap ? fd : fs, bf_slot(swap ? fd : fs, src_slot), + &s.csel1) || + sel_sop3(swap ? fs : fd, bf_slot(swap ? fs : fd, src_slot), + &s.csel2)) + return -ENOTSUP; + s.cmod1 = (swap ? fd : fs).inv; + s.cmod2 = (swap ? fs : fd).inv; + bl_put(l, enc_sopwm(&s)); + /* A, likewise, from the alpha lanes. */ + swap = arev; + src_slot = swap ? 2u : 1u; + s.s1bank = swap ? op2bank : op1bank; + s.s1 = swap ? op2 : op1; + s.s2bank = swap ? op1bank : op2bank; + s.s2 = swap ? op1 : op2; + s.cop = s.aop = aop; + s.mask = 0x1u; + as = bf_lane(as); + ad = bf_lane(ad); + if (sel_sop3(swap ? ad : as, bf_slot(swap ? ad : as, src_slot), + &s.csel1) || + sel_sop3(swap ? as : ad, bf_slot(swap ? as : ad, src_slot), + &s.csel2)) + return -ENOTSUP; + s.cmod1 = (swap ? ad : as).inv; + s.cmod2 = (swap ? as : ad).inv; + bl_put(l, enc_sopwm(&s)); + return l->err; +} + +/* Scale one operand by its factors into `dst`: the vendor's + * EncodeThreeSourceBlend(), which is how a constant reaches a form with two + * operand selects. The scaled operand sits in `slot`; the other operand of + * the SOPWM is whatever the factor refers to. One instruction when the alpha + * factor is the colour factor's own alpha lane, else RGB and A separately. */ +static int bl_scale(struct sgx_bl *l, struct sgx_bf f, struct sgx_bf fa, + unsigned slot, unsigned opbank, unsigned op, + unsigned src, unsigned rc, unsigned dst) +{ + struct sgx_sop s; + unsigned k, passes = 1; + struct sgx_bf fac[2]; + unsigned masks[2]; + + fac[0] = f; + masks[0] = 0xfu; + if (f.kind == SGX_BK_ASAT || !bf_same(bf_lane(f), bf_lane(fa))) { + passes = 2; + masks[0] = 0xeu; + fac[1] = bf_lane(fa); + masks[1] = 0x1u; + } + for (k = 0; k < passes; k++) { + unsigned obank, oreg, oslot = 3u - slot, sel; + + /* Scaling in place, an alpha of one has nothing to do. */ + if (k && dst == op && opbank == SGX_BANK_TEMP && + bf_is_one(fac[k])) + break; + + /* The other register: the constant, the source, or the + * destination in o0. Saturate wants the source in operand 1 + * and the destination in operand 2, whichever is scaled. */ + switch (fac[k].kind) { + case SGX_BK_CONST: obank = SGX_BANK_TEMP; oreg = rc; break; + case SGX_BK_DST: obank = SGX_BANK_OUTPUT; oreg = 0; break; + case SGX_BK_ASAT: + obank = slot == 1u ? SGX_BANK_OUTPUT : SGX_BANK_TEMP; + oreg = slot == 1u ? 0 : src; + break; + default: obank = SGX_BANK_TEMP; oreg = src; break; + } + memset(&s, 0, sizeof s); + s.dbank = SGX_BANK_TEMP; + s.dst = dst; + s.mask = masks[k]; + if (slot == 1u) { + s.s1bank = opbank; s.s1 = op; + s.s2bank = obank; s.s2 = oreg; + } else { + s.s1bank = obank; s.s1 = oreg; + s.s2bank = opbank; s.s2 = op; + } + /* A factor names the slot its referent sits in; saturate is + * min(src1.a, 1 - src2.a) in either arrangement. */ + if (fac[k].kind == SGX_BK_ASAT) + sel = 1; + else if (fac[k].kind == SGX_BK_ZERO) + sel = 0; + else if (sel_sop3(fac[k], fac[k].kind == SGX_BK_SRC ? 1u : + fac[k].kind == SGX_BK_DST ? 2u : oslot, + &sel)) + return -ENOTSUP; + if (slot == 1u) { + s.csel1 = sel; s.cmod1 = fac[k].inv; + } else { + s.csel2 = sel; s.cmod2 = fac[k].inv; + } + bl_put(l, enc_sopwm(&s)); + } + return l->err; +} + +/* The single-instruction constant form: src0 holds the constant, and the + * alpha destination factor is whatever CSEL2's alpha lane gives. */ +static int bl_sop3(struct sgx_bl *l, struct sgx_bf fs, struct sgx_bf fd, + struct sgx_bf as, struct sgx_bf ad, unsigned cop, + unsigned crev, unsigned aop, unsigned arev, + unsigned src, unsigned rc, unsigned dbank, unsigned dst) +{ + struct sgx_sop s; + unsigned swap = crev, src_slot = swap ? 2u : 1u; + struct sgx_bf f1 = swap ? fd : fs, f2 = swap ? fs : fd; + struct sgx_bf a1 = swap ? ad : as, a2 = swap ? as : ad; + + if (cop >= 2u || aop >= 2u || (aop == 1u && arev != swap)) + return -ENOTSUP; + /* The implicit alpha factor 2 is f2's lane; saturate's lane in this + * form is not established. */ + if (f2.kind == SGX_BK_ASAT || !bf_same(bf_lane(f2), bf_lane(a2))) + return -ENOTSUP; + memset(&s, 0, sizeof s); + s.dbank = dbank; + s.dst = dst; + s.s0 = rc; + s.s1bank = swap ? SGX_BANK_OUTPUT : SGX_BANK_TEMP; + s.s1 = swap ? 0 : src; + s.s2bank = swap ? SGX_BANK_TEMP : SGX_BANK_OUTPUT; + s.s2 = swap ? src : 0; + s.cop = cop; + s.aop = aop; + if (sel_sop3(f1, bf_slot(f1, src_slot), &s.csel1) || + sel_sop3(f2, bf_slot(f2, src_slot), &s.csel2) || + asel_sop3(a1, bf_slot(a1, src_slot), &s.asel1)) + return -ENOTSUP; + s.cmod1 = f1.inv; + s.cmod2 = f2.inv; + s.amod1 = bf_lane(a1).inv; + bl_put(l, enc_sop3(&s)); + return l->err; +} + +/* Whether the state names the fragment program's second colour. The caller + * has to have compiled one and to know which register it is in. */ +int sgx_blend_uses_src1(const struct sgx_blend_desc *d) +{ + struct sgx_bf a, b, c, e; + + if (!d || !d->enable) + return 0; + return (!bf_decode(d->rgb_src, &a) && a.kind == SGX_BK_SRC1) || + (!bf_decode(d->rgb_dst, &b) && b.kind == SGX_BK_SRC1) || + (!bf_decode(d->alpha_src, &c) && c.kind == SGX_BK_SRC1) || + (!bf_decode(d->alpha_dst, &e) && e.kind == SGX_BK_SRC1); +} + +int sgx_blend_build(const struct sgx_blend_desc *d, uint32_t constant, + unsigned src, unsigned src1, unsigned tmp, unsigned ntmp, + unsigned flags, uint64_t *out, unsigned cap, unsigned *n, + unsigned *used) +{ + struct sgx_bf fs, fd, as, ad; + struct sgx_bl l = { out, 0, cap, 0 }; + unsigned cop, crev, aop, arev, next = tmp; + unsigned rc = 0, rd = 0, rt = 0, has_const, has_src1; + unsigned op2bank = SGX_BANK_OUTPUT, op2 = 0; + unsigned dbank = SGX_BANK_OUTPUT, dst = 0; + int split = (flags & SGX_BLEND_F_SPLIT) != 0; + int r; + + if (!d || !out || !n || !used || src > 0x7fu) + return -EINVAL; + *n = 0; + *used = 0; + /* A write mask without blending is the copy alone. */ + if (!d->enable) { + struct sgx_sop s; + + if (d->mask == 0xfu) + return -EINVAL; + memset(&s, 0, sizeof s); + s.dbank = SGX_BANK_OUTPUT; + s.s1bank = s.s2bank = SGX_BANK_TEMP; + s.s1 = s.s2 = src; + s.mask = wm_nibble(d->mask); + s.cmod1 = 1; /* one, zero: the source */ + bl_put(&l, enc_sopwm(&s) | ((uint64_t)SGX_USE1_END << 32)); + *n = l.n; + return l.err; + } + if (bf_decode(d->rgb_src, &fs) || bf_decode(d->rgb_dst, &fd) || + bf_decode(d->alpha_src, &as) || bf_decode(d->alpha_dst, &ad) || + blend_op(d->rgb_func, &cop, &crev) || + blend_op(d->alpha_func, &aop, &arev)) + return -ENOTSUP; + /* MIN and MAX take no factors: the vendor writes ONE/ONE, and GL + * says the factors are ignored. Per side, since each has its own. */ + if (cop >= 2u) + fs = fd = bf_one; + if (aop >= 2u) + as = ad = bf_one; + has_const = fs.kind == SGX_BK_CONST || fd.kind == SGX_BK_CONST || + as.kind == SGX_BK_CONST || ad.kind == SGX_BK_CONST; + has_src1 = fs.kind == SGX_BK_SRC1 || fd.kind == SGX_BK_SRC1 || + as.kind == SGX_BK_SRC1 || ad.kind == SGX_BK_SRC1; + /* Both want SOP3's src0 and only one can have it. GL allows the pair; + * the part cannot name three colours and a constant in one + * instruction, and staging the constant into a temporary would need a + * fourth operand this form has no slot for. */ + if (has_const && has_src1) + return -ENOTSUP; + if (has_src1) { + if (src1 > 0x7fu) + return -EINVAL; + /* The second colour is already in a register - the fragment + * program computed it - so src0 names it rather than staging + * anything. */ + rc = src1; + } else if (has_const) { + if (next - tmp >= ntmp) + return -ENOSPC; + rc = next++; + bl_put(&l, enc_limm(rc, constant)); + } + if (d->mask != 0xfu) { + if (next - tmp >= ntmp) + return -ENOSPC; + rt = next++; + dbank = SGX_BANK_TEMP; + dst = rt; + } + if (!has_const && !has_src1) { + r = bl_equation(&l, fs, fd, as, ad, cop, crev, aop, arev, + SGX_BANK_TEMP, src, SGX_BANK_OUTPUT, 0, + dbank, dst, split); + } else if (!split && + !bl_sop3(&l, fs, fd, as, ad, cop, crev, aop, arev, src, + rc, dbank, dst)) { + r = 0; + } else if (has_src1) { + /* The staged form below scales through src0, which is where + * the second colour is; there is nowhere to put it. A dual + * source blend is the one instruction or nothing. */ + return -ENOTSUP; + } else { + /* Scale both operands by their factors first - the + * destination before the source, which it may read - and + * combine them with ONE/ONE. */ + if (!bf_is_one(fd) || !bf_is_one(ad)) { + if (next - tmp >= ntmp) + return -ENOSPC; + rd = next++; + r = bl_scale(&l, fd, ad, 2u, SGX_BANK_OUTPUT, 0, src, + rc, rd); + if (r) + return r; + op2bank = SGX_BANK_TEMP; + op2 = rd; + } + if (!bf_is_one(fs) || !bf_is_one(as)) { + r = bl_scale(&l, fs, as, 1u, SGX_BANK_TEMP, src, src, + rc, src); + if (r) + return r; + } + r = bl_equation(&l, bf_one, bf_one, bf_one, bf_one, cop, crev, + aop, arev, SGX_BANK_TEMP, src, op2bank, op2, + dbank, dst, split); + } + if (r) + return r; + if (d->mask != 0xfu) { + /* The vendor's CreateColorMaskUSECode(): the masked channels + * are copied into o0 by a write-masked instruction. */ + struct sgx_sop s; + + memset(&s, 0, sizeof s); + s.dbank = SGX_BANK_OUTPUT; + s.s1bank = s.s2bank = SGX_BANK_TEMP; + s.s1 = s.s2 = rt; + s.mask = wm_nibble(d->mask); + s.cmod1 = 1; + bl_put(&l, enc_sopwm(&s)); + } + if (l.err) + return l.err; + out[l.n - 1u] |= (uint64_t)SGX_USE1_END << 32; + *n = l.n; + *used = next - tmp; + return 0; +} + +int sgx_blend_insn(unsigned rgb_src, unsigned rgb_dst, unsigned alpha_src, + unsigned alpha_dst, unsigned rgb_func, unsigned alpha_func, + uint32_t *hi, uint32_t *lo) +{ + struct sgx_blend_desc d; + uint64_t insn[SGX_BLEND_MAX_INSNS]; + unsigned n = 0, used = 0; + + if (!hi || !lo) + return -1; + memset(&d, 0, sizeof d); + d.rgb_src = (unsigned char)rgb_src; + d.rgb_dst = (unsigned char)rgb_dst; + d.alpha_src = (unsigned char)alpha_src; + d.alpha_dst = (unsigned char)alpha_dst; + d.rgb_func = (unsigned char)rgb_func; + d.alpha_func = (unsigned char)alpha_func; + d.mask = 0xf; + d.enable = 1; + if (sgx_blend_build(&d, 0, 0, 0, 1, 1, 0, insn, SGX_BLEND_MAX_INSNS, + &n, &used) || n != 1u || + ((insn[0] >> 59) & 0x1fu) != 0x10u) + return -1; + *hi = (uint32_t)(insn[0] >> 32); + *lo = (uint32_t)insn[0]; + return 0; +} + +uint32_t sgx_blend_const_pack(const float *rgba) +{ + static const unsigned char shift[4] = { 16, 8, 0, 24 }; + uint32_t w = 0; + unsigned i; + + for (i = 0; i < 4; i++) { + float c = rgba[i]; + unsigned b = c <= 0.0f ? 0u : c >= 1.0f ? 255u : + (unsigned)(c * 255.0f + 0.5f); + + w |= b << shift[i]; + } + return w; +} + +/* GL logic ops, as the part does them: a bitwise instruction at the end of the + * fragment program with the destination register o0 as one operand. Under the + * read-modify-write object type - the same one blending uses - o0 arrives + * holding the pixel already in the framebuffer, and what the program leaves + * there is written back, so "xor o0, colour, o0" is the whole of GL_XOR. + * + * The bitwise group is w1[31:27]: 0x0a is AND with w1[3] clear and OR with it + * set, 0x0b is XOR. w1[11] complements source 2. Operands are the packed + * 8888 dword, so one 32-bit operation is the per-channel one GL asks for. + * + * The words are the vendor's own, read out of psb_dri.so's logic-op emitter + * (the jump table at .rodata 0x259348, indexed by the op less GL_CLEAR) and + * checked field by field against USSE_ISA.txt. Two of the sixteen take a + * second instruction to invert the result. + * + * c56 reads all-ones. That is the one part of this not measured: the constant + * bank's 0.0 and 1.0 groups are established, the 16-bit-plus-rotate immediate + * form cannot express 0xffffffff, and the vendor uses c56 wherever it needs a + * NOT - so it can be nothing else. GL_INVERT passing is the proof. + */ +#define SGX_LOP_SRC_NONE 0 /* the op does not read the colour */ +#define SGX_LOP_SRC_1 1 /* source 1, w0[13:7] */ +#define SGX_LOP_SRC_2 2 /* source 2, w0[6:0] */ +#define SGX_LOP_SRC_BOTH 3 /* both, which is how COPY is done */ + +static const struct { + uint32_t hi, lo; /* w1 and w0, with the colour left out */ + unsigned char where; /* where the colour register goes */ + uint32_t hi2, lo2; /* a second instruction, or zero */ +} sgx_lop[16] = { + /* PIPE_LOGICOP_CLEAR xor o0, o0, o0 */ + { 0x58a00001u, 0x50000000u, SGX_LOP_SRC_NONE, 0, 0 }, + /* NOR or o0, s, o0 ; xor o0, c56, o0 */ + { 0x50a00009u, 0x10000000u, SGX_LOP_SRC_1, + 0x58a20001u, 0x50001c00u }, + /* AND_INVERTED and o0, o0, ~s */ + { 0x50a00801u, 0x40000000u, SGX_LOP_SRC_2, 0, 0 }, + /* COPY_INVERTED xor o0, c56, s */ + { 0x58a20001u, 0x40001c00u, SGX_LOP_SRC_2, 0, 0 }, + /* AND_REVERSE and o0, s, ~o0 */ + { 0x50a00801u, 0x10000000u, SGX_LOP_SRC_1, 0, 0 }, + /* INVERT xor o0, c56, o0 */ + { 0x58a20001u, 0x50001c00u, SGX_LOP_SRC_NONE, 0, 0 }, + /* XOR xor o0, s, o0 */ + { 0x58a00001u, 0x10000000u, SGX_LOP_SRC_1, 0, 0 }, + /* NAND and o0, s, o0 ; xor o0, c56, o0 */ + { 0x50a00001u, 0x10000000u, SGX_LOP_SRC_1, + 0x58a20001u, 0x50001c00u }, + /* AND and o0, s, o0 */ + { 0x50a00001u, 0x10000000u, SGX_LOP_SRC_1, 0, 0 }, + /* EQUIV xor o0, s, ~o0 */ + { 0x58a00801u, 0x10000000u, SGX_LOP_SRC_1, 0, 0 }, + /* NOOP or o0, o0, o0 */ + { 0x50a00009u, 0x50000000u, SGX_LOP_SRC_NONE, 0, 0 }, + /* OR_INVERTED or o0, o0, ~s */ + { 0x50a00809u, 0x40000000u, SGX_LOP_SRC_2, 0, 0 }, + /* COPY or o0, s, s */ + { 0x50a00009u, 0x00000000u, SGX_LOP_SRC_BOTH, 0, 0 }, + /* OR_REVERSE or o0, s, ~o0 */ + { 0x50a00809u, 0x10000000u, SGX_LOP_SRC_1, 0, 0 }, + /* OR or o0, s, o0 */ + { 0x50a00009u, 0x10000000u, SGX_LOP_SRC_1, 0, 0 }, + /* SET xor o0, o0, ~o0 */ + { 0x58a00801u, 0x50000000u, SGX_LOP_SRC_NONE, 0, 0 }, +}; + +int sgx_logicop_insn(unsigned func, unsigned reg, uint64_t *out, unsigned *n) +{ + uint32_t lo; + + if (func >= 16 || !out || !n || reg > 0x7fu) + return -1; + lo = sgx_lop[func].lo; + if (sgx_lop[func].where & SGX_LOP_SRC_1) + lo |= (reg & 0x7fu) << 7; + if (sgx_lop[func].where & SGX_LOP_SRC_2) + lo |= reg & 0x7fu; + out[0] = ((uint64_t)sgx_lop[func].hi << 32) | lo; + *n = 1; + if (sgx_lop[func].hi2) { + out[1] = ((uint64_t)sgx_lop[func].hi2 << 32) | + sgx_lop[func].lo2; + *n = 2; + } + /* The vendor's words carry no .end - its emitter appends a terminator + * of its own afterwards. These replace the move that ended the + * program, so the last of them has to end it, or the USSE runs on + * into whatever follows and the pixel is never written. */ + out[*n - 1] |= (uint64_t)(1u << 18) << 32; + return 0; +} diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_state.h mesa-26.2.2/src/gallium/drivers/sgx/sgx_state.h --- mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_state.h 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/sgx_state.h 2026-09-08 10:57:36.681241988 +0200 @@ -0,0 +1,293 @@ +/* SPDX-License-Identifier: MIT */ +/* + * Gallium state to SGX535 ISP words. + * + * This is the part of a Gallium driver that the reverse engineering actually + * pays for: turning pipe_depth_stencil_alpha_state and pipe_rasterizer_state + * into the per-draw words the hardware reads. The field layout is + * GL-IMPLEMENTATION.md section 3, "Per-draw state", verified there against + * enum sweeps on hardware. + * + * It takes plain structs rather than Gallium's, so it builds and is tested + * without a Mesa tree; the field names match Gallium's so wiring it up later is + * assignment rather than translation. + * + * Copyright (C) 2026 René Rebe + */ +#ifndef _SGX_STATE_H_ +#define _SGX_STATE_H_ + +#include + +/* Gallium's PIPE_FUNC_* order, which is GL's minus GL_NEVER, and the same + * order the hardware's compare field uses - so this one needs no mapping. */ +enum sgx_func { + SGX_FUNC_NEVER = 0, SGX_FUNC_LESS, SGX_FUNC_EQUAL, SGX_FUNC_LEQUAL, + SGX_FUNC_GREATER, SGX_FUNC_NOTEQUAL, SGX_FUNC_GEQUAL, SGX_FUNC_ALWAYS, +}; + +/* Gallium's PIPE_STENCIL_OP_* order. The hardware's is different - INVERT is 5 + * there and 7 here, and the two WRAP ops move with it - so this one does need + * a mapping, and getting it wrong is silent: the wrong op still renders. */ +enum sgx_stencil_op { + SGX_STENCIL_KEEP = 0, SGX_STENCIL_ZERO, SGX_STENCIL_REPLACE, + SGX_STENCIL_INCR, SGX_STENCIL_DECR, SGX_STENCIL_INCR_WRAP, + SGX_STENCIL_DECR_WRAP, SGX_STENCIL_INVERT, +}; + +struct sgx_stencil_face { + unsigned enabled; + unsigned func; /* enum sgx_func */ + unsigned fail_op, zfail_op, zpass_op; + unsigned valuemask, writemask; +}; + +struct sgx_dsa_state { + unsigned depth_enabled; + unsigned depth_writemask; + unsigned depth_func; /* enum sgx_func */ + unsigned stencil_ref; + unsigned stencil_ref_back; /* pipe_stencil_ref::ref_value[1] */ + struct sgx_stencil_face stencil[2]; +}; + +struct sgx_rasterizer_state { + unsigned cull_face; /* 0 none, 1 front, 2 back */ + unsigned front_ccw; + unsigned flatshade; + unsigned flatshade_first; /* provoking vertex first, else last */ + unsigned target_y_flipped; + /* Whether the part ran the vertex transform. Clear on the ordinary + * path, where the draw module transformed on the CPU - which is what + * bit 16 of the cull word reports. */ + unsigned hw_transform; + /* glPolygonOffset, pipe_rasterizer_state's offset_tri, offset_units + * and offset_scale. Only the part's transform applies it - the draw + * module adds the bias itself on the CPU path. */ + unsigned offset_tri; + float offset_units, offset_scale; +}; + +/* One ISP state: word A, and B and C when A says they follow (the DDK's + * EURASIA_ISPA_BPRES and CPRES, sgxdefs.h:1377-1378). Two of them under + * EURASIA_ISPA_2SIDED (:1375): the front-face set and the back-face set. */ +struct sgx_isp_set { + uint32_t a, b, c; +}; + +#define SGX_ISP_2SIDED (1u << 11) +#define SGX_ISP_BPRES (1u << 9) +#define SGX_ISP_CPRES (1u << 8) +/* Word B on this core (sgxdefs.h:1420-1438, the non-545 branch): the depth + * bias factor at [24:20] and units at [19:15], each a signed five-bit field, + * the user pass-spawn control at [7:4] and the visibility test at [3:0]. */ +#define SGX_ISPB_DBIASFACTOR_SHIFT 20 +#define SGX_ISPB_DBIASUNITS_SHIFT 15 +#define SGX_ISPB_UPASSCTL_SHIFT 4 +#define SGX_ISPB_UPASSCTL_MASK (0xfu << SGX_ISPB_UPASSCTL_SHIFT) +/* The visibility test: enable at bit 3, EURASIA_ISPB_VISTEST (sgxdefs.h:1435), + * and which counter it increments at [2:0], EURASIA_ISPB_VISREG (:1437-1438). + * Three bits, so eight counters and no ninth - EUR_CR_ISP_VISTEST_VISIBLE0..7 + * (sgx535defs.h:1378-1415). The in-memory form that splits the index across + * both words is another core's (:1360-1400); this one has neither that nor + * EURASIA_ISPA_VISBOOL (:1345). */ +#define SGX_ISPB_VISTEST (1u << 3) +#define SGX_ISPB_VISREG_MASK 7u +#define SGX_ISP_VISTEST_REGS 8u +/* What the vendor sends for every draw with nothing set: all four channels + * (opengles2/validate.c:3609). A record whose word is this needs no B. */ +#define SGX_ISPB_DEFAULT (0xfu << SGX_ISPB_UPASSCTL_SHIFT) +/* Word C with the test off: ALWAYS and KEEP everywhere (sgxdefs.h:1451, + * 1470-1500). The vendor sends no C for it (validate.c:3887-3890). */ +#define SGX_ISPC_DEFAULT (7u << 25) + +/* The two ISP words. Word 0 is always present; word 2 only when the stencil + * test is on, which is what bit 8 of word 0 announces. */ +/* Blending, as the hardware does it: not a fixed-function unit but a pair of + * SOP2 instructions at the end of the fragment program, with bit 25 of the ISP + * word telling the pixel back end to read the destination. What can be + * expressed is the set of Render operators the shader generator emits, so a + * Gallium blend state maps onto one of those or onto none. */ +enum sgx_blend_op { + SGX_BLEND_NONE = -1, /* the caller's program writes the pixel */ + SGX_BLEND_CLEAR = 0, + SGX_BLEND_SRC = 1, + SGX_BLEND_DST = 2, + SGX_BLEND_OVER = 3, /* src + dst * (1 - src alpha), premultiplied */ + SGX_BLEND_ADD = 12, + SGX_BLEND_SATURATE = 13 +}; + +/* What a blend asks for, in Gallium's numbering. The instructions are built + * from it by sgx_blend_build() once the registers are known, so it is what a + * bound state and a frame's record both carry. */ +struct sgx_blend_desc { + unsigned char rgb_src, rgb_dst, alpha_src, alpha_dst; /* PIPE_BLENDFACTOR_* */ + unsigned char rgb_func, alpha_func; /* PIPE_BLEND_* */ + unsigned char mask; /* PIPE_MASK_*, 0xf when nothing is masked */ + unsigned char enable; /* blending; else a masked copy alone */ +}; + +struct sgx_blend_state { + enum sgx_blend_op op; + /* The blend this state compiles to, when it can be built from the + * factors rather than matched against a named operator. */ + struct sgx_blend_desc desc; + int have_insn; + unsigned colormask; /* PIPE_MASK_*, 0xf when nothing is masked */ + /* GL_COLOR_LOGIC_OP: the op replaces the blend equation, and only the + * three the blend unit can express are carried out. */ + unsigned logicop; + unsigned logicop_func; + int logicop_ok; + /* Whether the ISP has to treat the object as translucent, which is not + * the same question as whether it blends: see sgx_blend_translucent(). + * A write mask or a logic op still reads the destination back and is + * still sorted as an opaque object. */ + int translucent; +}; + +/* Gallium's blend factor codes, for the places that classify a blend rather + * than encode it. Named here because sgx_state.c is built without the pipe + * headers so it can be tested on the host. */ +#define SGX_BF_ZERO 0x11u +#define SGX_BF_DST_ALPHA 0x04u +#define SGX_BF_DST_COLOR 0x05u +#define SGX_BF_SRC_ALPHA_SAT 0x06u +#define SGX_BF_INV_DST_ALPHA 0x14u +#define SGX_BF_INV_DST_COLOR 0x15u + +/* The vendor's IsBlendTranslucent(): the destination survives the blend, so + * the part may not remove the object before shading it. True when either + * destination factor is not ZERO, or a source factor reads the destination. + * Everything else - a write mask, a logic op, a blend that discards what was + * there - stays opaque and keeps hidden surface removal. */ +int sgx_blend_translucent(unsigned rgb_src, unsigned rgb_dst, + unsigned alpha_src, unsigned alpha_dst); + +/* Build the instructions that carry out a blend, into out[] (room for + * SGX_BLEND_MAX_INSNS). The equation lives in the fragment program on this + * part, so blending is a final instruction rather than a register - see + * GL-IMPLEMENTATION.md, "Blending - decoded; the equation lives in the + * shader". `src` is the temporary holding the packed source colour, the + * destination is o0; temporaries from `tmp` up (`ntmp` of them) are for the + * constant, a pre-scaled destination and the masked result, and *used says + * how many were taken. The last instruction carries .end. Returns 0, or + * -ENOTSUP for what no form expresses and -ENOSPC when out of room. */ +#define SGX_BLEND_MAX_INSNS 8u +/* Carry a mixed equation as two write-masked instructions (RGB, then A) + * rather than one SOP2 with per-side selects - the alternative form, for + * the hardware to decide between. */ +#define SGX_BLEND_F_SPLIT 1u +/* `src1` is the register holding the fragment program's second colour, for a + * state whose factors name it; ignored otherwise. It reaches the blend as + * SOP3's src0, so a dual-source state cannot also carry a blend constant and + * is refused with -ENOTSUP rather than encoded wrongly. */ +int sgx_blend_build(const struct sgx_blend_desc *d, uint32_t constant, + unsigned src, unsigned src1, unsigned tmp, unsigned ntmp, + unsigned flags, uint64_t *out, unsigned cap, unsigned *n, + unsigned *used); +/* Whether any of a state's factors is a dual-source one, so the caller knows + * it has to have a second colour to give. */ +int sgx_blend_uses_src1(const struct sgx_blend_desc *d); + +/* The single SOP2 for a full-mask blend without a constant, its source + * register left as zero: what the host checks and the logic-op path use. */ +int sgx_blend_insn(unsigned rgb_src, unsigned rgb_dst, unsigned alpha_src, + unsigned alpha_dst, unsigned rgb_func, unsigned alpha_func, + uint32_t *hi, uint32_t *lo); + +/* The low word with S1BANK [31:30] output and S2BANK [29:28] temporary - the + * REVERSE_SUBTRACT swap; the colour register is then SRC2 [6:0], o0 operand 1. */ +#define SGX_BLEND_LO_SWAPPED(lo) (((lo) & 0xc0000000u) == 0x40000000u) + +/* The blend constant as the part reads it: packed A[31:24] R[23:16] G[15:8] + * B[7:0], each round(c * 255) - opengles2/misc.c ColorConvertToHWFormat(). */ +uint32_t sgx_blend_const_pack(const float *rgba); + +/* PIPE_LOGICOP_COPY, in Mesa's numbering: the identity, and the one logic op + * that needs neither the destination nor an instruction of its own. */ +#define SGX_LOGICOP_COPY 12u + +/* Build the instructions for one GL logic op into out[], which must have room + * for two. `reg` is the register holding the fragment's packed colour; the + * destination is o0, which under the read-modify-write object type arrives + * holding the pixel already there. Returns 0 and sets *n, or non-zero when + * the op is not one of the sixteen. */ +int sgx_logicop_insn(unsigned func, unsigned reg, uint64_t *out, unsigned *n); + +uint32_t sgx_isp_word0(const struct sgx_dsa_state *dsa); +/* The same for one face of a two-sided state: that face's reference, and + * CPRES when its word C is not the default. */ +uint32_t sgx_isp_word0_face(const struct sgx_dsa_state *dsa, unsigned face); +/* Which face a single set describes, by the vendor's rule + * (opengles2/validate.c:3722-3737): the one a cull leaves. Without a cull the + * two faces get a set each - see sgx_isp_sets(). */ +unsigned sgx_isp_stencil_face(const struct sgx_dsa_state *dsa, + const struct sgx_rasterizer_state *r); +uint32_t sgx_isp_word2(const struct sgx_stencil_face *f); +/* Word B: the depth bias, only when the part transforms, and the write mask + * as the pass-spawn key. colormask is Mesa's PIPE_MASK_* set. */ +uint32_t sgx_isp_word1(const struct sgx_rasterizer_state *r, int hw_transform, + unsigned colormask); +/* Whether the face the ISP calls front under 2SIDED - clockwise in device + * space - is GL's front face. */ +int sgx_isp_front_is_gl_front(const struct sgx_rasterizer_state *r); +/* Arm an object for one of the eight visibility counters. Both fields sit in + * word B, so the set has to announce a B rather than append one - the vendor's + * rule that a full word B goes with every object that needs any of it + * (validate.c:3884-3885). A two-sided set counts both faces into the same + * register. reg < 0 leaves the set alone; -EINVAL for an index the three bits + * cannot name. */ +int sgx_isp_set_vistest(struct sgx_isp_set *ff, struct sgx_isp_set *bf, + int reg); +/* The whole per-draw ISP state: the front set, and under 2SIDED the back + * set, from the depth/stencil state, the rasterizer state and the colour + * write mask. The object type, the pass type and the blend bit are the + * context's to add. */ +void sgx_isp_sets(const struct sgx_dsa_state *dsa, + const struct sgx_rasterizer_state *r, int hw_transform, + unsigned colormask, struct sgx_isp_set *ff, + struct sgx_isp_set *bf); + +/* The object the ISP rasterises, ISP word A bits [18:15] - the DDK's + * EURASIA_ISPA_OBJTYPE. A triangle is zero, which is what every frame this + * driver has ever built carries. The vendor maps GL points to SPRITEUV and + * lines to LINE; SPRITE01UV is the form that also generates the sprite + * coordinate, for gl_PointCoord. */ +#define SGX_ISP_OBJTYPE_SHIFT 15 +#define SGX_ISP_OBJTYPE_MASK (0xfu << SGX_ISP_OBJTYPE_SHIFT) +#define SGX_ISP_OBJ_TRI 0u +#define SGX_ISP_OBJ_LINE 1u +#define SGX_ISP_OBJ_SPRITE10UV 2u +#define SGX_ISP_OBJ_SPRITEUV 3u +#define SGX_ISP_OBJ_SPRITE01UV 4u +#define SGX_ISP_OBJ_LINETRI 5u +#define SGX_ISP_OBJ_POINTTRI 6u + +/* Point and line width, ISP word A bits [31:28], as width - 1. Four bits, so + * the widest the hardware rasterises is sixteen. */ +#define SGX_ISP_PLWIDTH_SHIFT 28 +#define SGX_ISP_PLWIDTH_MASK (0xfu << SGX_ISP_PLWIDTH_SHIFT) +#define SGX_ISP_PLWIDTH_MAX 16u + +/* Put an object type and a width into an ISP word A. width is in pixels and + * is only read for a point or a line; it is clamped to what the field can + * describe rather than refused. */ +uint32_t sgx_isp_objtype(uint32_t word0, unsigned objtype, unsigned width); +uint32_t sgx_raster_word(const struct sgx_rasterizer_state *r); + +/* The MTE control word's shade model, [18:17], as the DDK's + * EURASIA_MTE_SHADE_* on this core (sgxdefs.h:2361-2366): 0 gouraud, then + * 1 + the corner a flat value comes from. Three units read the same corner - + * the MTE here, the iterator's own DOUTI FLATSHADE field and the vertex data + * master's index list - and the vendor writes all three together, so this is + * what the other two are derived from. */ +#define SGX_MTE_SHADE_SHIFT 17 +#define SGX_MTE_SHADE_MASK (3u << SGX_MTE_SHADE_SHIFT) +unsigned sgx_raster_flat_corner(uint32_t raster_word); + +/* Does this state need the second ISP word emitted? */ +int sgx_isp_has_stencil(const struct sgx_dsa_state *dsa); + +#endif diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_twod.h mesa-26.2.2/src/gallium/drivers/sgx/sgx_twod.h --- mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_twod.h 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/sgx_twod.h 2026-09-08 10:57:36.681318122 +0200 @@ -0,0 +1,389 @@ +/* SPDX-License-Identifier: GPL-2.0-only */ +/* + * The SGX535 2D block: command encoding, and the check a submitted stream + * has to pass before the kernel writes it to the slave port. + * + * The engine takes variable-length "block header" commands as dwords through + * PSB_SGX_2D_SLAVE_PORT, and addresses memory through the BIF as offsets from + * BIF_TWOD_REQ_BASE. Every encoding below is transcribed from psb_reg.h of the + * GPL psb-kernel-source (kept in this tree as driver/gma500/psb_reg.h) and + * cross-checked against xserver-xorg-video-psb-0.32.1/src/psb_accel.c, which + * drove this engine on this silicon, and against tools/baremetal/sgxtri.c + * stage twod, which ran it again here (work/twod-cursor/README.md). + * + * Pure logic, so the module and a host test share it verbatim: the checker is + * a security boundary - a stream it passes reaches memory through the client's + * own page tables - and it is tested before the hardware is. + * + * Copyright (C) 2026 René Rebe + */ +#ifndef _SGX_TWOD_H_ +#define _SGX_TWOD_H_ + +/* Registers. psb_drv.h:63, psb_reg.h:152-161 */ +#define SGX_2D_SLAVE_PORT 0x4000u +#define SGX_2D_SOCIF 0x0e18u +#define SGX_2D_SOCIF_FREE_MASK 0x000000ffu /* dwords the FIFO takes */ +#define SGX_2D_SOCIF_EMPTY 0x00000080u +#define SGX_2D_BLIT_STATUS 0x0e04u +#define SGX_2D_STATUS_BUSY 0x01000000u +#define SGX_2D_STATUS_COMPLETE_MASK 0x00ffffffu /* FENCE and FLUSH blocks retired */ +/* psb_sgx.c:160 - at most this many dwords between free-space checks. */ +#define SGX_2D_FIFO_CHUNK 0x60u +/* EVENT_STATUS bit 27, psb_reg.h:74. */ +#define SGX_2D_EV_COMPLETE 0x08000000u + +/* Block headers, psb_reg.h:173-188: object type in bits [31:28]. */ +#define SGX_2D_BH_MASK 0xf0000000u +#define SGX_2D_CLIP_BH 0x00000000u +#define SGX_2D_PAT_BH 0x10000000u +#define SGX_2D_CTRL_BH 0x20000000u +#define SGX_2D_SRC_OFF_BH 0x30000000u +#define SGX_2D_MASK_OFF_BH 0x40000000u +#define SGX_2D_FENCE_BH 0x70000000u +#define SGX_2D_BLIT_BH 0x80000000u +#define SGX_2D_SRC_SURF_BH 0x90000000u +#define SGX_2D_DST_SURF_BH 0xa0000000u +#define SGX_2D_PAT_SURF_BH 0xb0000000u +#define SGX_2D_SRC_PAL_BH 0xc0000000u +#define SGX_2D_PAT_PAL_BH 0xd0000000u +#define SGX_2D_MASK_SURF_BH 0xe0000000u +#define SGX_2D_FLUSH_BH 0xf0000000u + +/* Surface words, psb_reg.h:380-408 (source) and :450-469 (destination): the + * format in bits [18:15], the byte stride in [14:0], then one dword carrying + * the byte offset from BIF_TWOD_REQ_BASE in bits [27:2]. The same format + * codes serve both ends for the RGB layouts. */ +#define SGX_2D_SURF_FORMAT_MASK 0x00078000u +#define SGX_2D_SURF_STRIDE_MASK 0x00007fffu +#define SGX_2D_SURF_RESERVED 0x0ff80000u +#define SGX_2D_ADDR_MASK 0x0ffffffcu +#define SGX_2D_REACH 0x10000000u /* 28-bit offset field */ +#define SGX_2D_FMT_332RGB 0x00030000u +#define SGX_2D_FMT_4444ARGB 0x00038000u +#define SGX_2D_FMT_555RGB 0x00040000u +#define SGX_2D_FMT_1555ARGB 0x00048000u +#define SGX_2D_FMT_565RGB 0x00050000u +#define SGX_2D_FMT_0888ARGB 0x00058000u +#define SGX_2D_FMT_8888ARGB 0x00060000u + +/* Source offset, psb_reg.h:276-279: x in [23:12], y in [11:0]. The blit's + * start and size words use the same split, psb_reg.h:355-371. */ +#define SGX_2D_XY_MASK 0x00ffffffu +#define SGX_2D_COORD_MAX 0xfffu +#define SGX_2D_XY(x, y) (((unsigned int)(x) << 12) | (unsigned int)(y)) + +/* The blit command word, psb_reg.h:297-341. */ +#define SGX_2D_ROT_MASK (3u << 25) +#define SGX_2D_ROT_NONE (0u << 25) +#define SGX_2D_COPYORDER_MASK (3u << 23) +#define SGX_2D_COPYORDER_TL2BR (0u << 23) +#define SGX_2D_COPYORDER_BR2TL (1u << 23) +#define SGX_2D_COPYORDER_TR2BL (2u << 23) +#define SGX_2D_COPYORDER_BL2TR (3u << 23) +#define SGX_2D_DSTCK_MASK 0x00600000u +#define SGX_2D_SRCCK_MASK 0x00180000u +#define SGX_2D_CLIP_ENABLE 0x00040000u +#define SGX_2D_ALPHA_ENABLE 0x00020000u +#define SGX_2D_USE_PAT 0x00010000u +#define SGX_2D_USE_FILL 0x00000000u +#define SGX_2D_ROP3B_SHIFT 8 +#define SGX_2D_ROP3A_SHIFT 0 +#define SGX_2D_BLIT_RESERVED 0x08000000u +/* work/twod-cursor/README.md: with only ROP3A set the blit is silently + * consumed - the engine waits on a mask surface that is never bound. */ +#define SGX_2D_ROP3(op) (((unsigned int)(op) << SGX_2D_ROP3A_SHIFT) | \ + ((unsigned int)(op) << SGX_2D_ROP3B_SHIFT)) +#define SGX_2D_ROP_SRCCOPY 0xccu +#define SGX_2D_ROP_PATCOPY 0xf0u + +/* One blit and its state, and the largest stream a submit may carry. Nine + * dwords per copy - both surfaces re-described, offset, four for the blit - + * gives a display server 1800 rectangles per ioctl. */ +#define SGX_2D_MAX_DWORDS 16384u +/* What the kernel appends: a surface, a 1x1 fill of the sequence number + * into its own page, FENCE and FLUSH. */ +#define SGX_2D_TRAILER_DWORDS 8u + +/* Bytes per pixel of a surface word's format, or 0 for a format the checker + * does not let through: palettes and alpha-only need a PAL or MASK block + * that is refused below, and the YUV layouts have no measured extent. */ +static inline unsigned int sgx_2d_format_bpp(unsigned int fmt) +{ + switch (fmt & SGX_2D_SURF_FORMAT_MASK) { + case SGX_2D_FMT_332RGB: + return 1; + case SGX_2D_FMT_4444ARGB: + case SGX_2D_FMT_555RGB: + case SGX_2D_FMT_1555ARGB: + case SGX_2D_FMT_565RGB: + return 2; + case SGX_2D_FMT_0888ARGB: + case SGX_2D_FMT_8888ARGB: + return 4; + default: + return 0; + } +} + +/* Whether a ROP3 reads the source. The code's bit index is (P<<2)|(S<<1)|D + * - SRCCOPY 0xcc sets exactly the S=1 bits - so it depends on S when the + * S=1 half differs from the S=0 half. */ +static inline int sgx_2d_rop_reads_src(unsigned int rop) +{ + return ((rop >> 2) & 0x33u) != (rop & 0x33u); +} + +static inline unsigned int sgx_2d_surf_word(unsigned int bh, unsigned int fmt, + unsigned int stride) +{ + return bh | (fmt & SGX_2D_SURF_FORMAT_MASK) | + (stride & SGX_2D_SURF_STRIDE_MASK); +} + +/* + * A binding lookup the checker asks for every extent a blit touches: nonzero + * if [va, va + size) lies wholly inside memory the caller may reach. The kernel + * answers from the file's surface bindings; a test answers from a table. + */ +typedef int (*sgx_2d_range_fn)(void *priv, unsigned long long va, + unsigned long long size); + +struct sgx_2d_report { + unsigned int at; /* dword index of the offending word */ + const char *why; +}; + +struct sgx_2d_surf_state { + unsigned int fmt, bpp, stride; + unsigned long long va; /* base + offset, once set */ + int set; +}; + +/* The rectangle a blit walks, in pixels: for a copy order that runs backwards + * the start names the far corner (pvr_copy() in xf86-video-gma500, + * psbExaSuperCopy() at psb_accel.c:1000-1005). Returns 0 if it does not fit + * the surface's rows or the 12-bit field. */ +static inline int sgx_2d_rect(unsigned int order, unsigned int x, + unsigned int y, unsigned int w, unsigned int h, + unsigned int *x0, unsigned int *y0, + unsigned int *x1, unsigned int *y1) +{ + int xback = order == SGX_2D_COPYORDER_BR2TL || + order == SGX_2D_COPYORDER_TR2BL; + int yback = order == SGX_2D_COPYORDER_BR2TL || + order == SGX_2D_COPYORDER_BL2TR; + + if (!w || !h || w > SGX_2D_COORD_MAX + 1u || h > SGX_2D_COORD_MAX + 1u) + return 0; + if (xback) { + if (x < w - 1u) + return 0; + *x0 = x - (w - 1u); + *x1 = x; + } else { + if (x + (w - 1u) > SGX_2D_COORD_MAX) + return 0; + *x0 = x; + *x1 = x + (w - 1u); + } + if (yback) { + if (y < h - 1u) + return 0; + *y0 = y - (h - 1u); + *y1 = y; + } else { + if (y + (h - 1u) > SGX_2D_COORD_MAX) + return 0; + *y0 = y; + *y1 = y + (h - 1u); + } + return 1; +} + +/* The rows a rectangle spans, whole, held against the caller's bindings. Whole + * rows rather than the pixels alone, because what the engine prefetches along + * a row is not documented; every surface this driver allocates is a whole + * number of rows. */ +static inline int sgx_2d_extent_ok(const struct sgx_2d_surf_state *s, + unsigned int x1, unsigned int y0, + unsigned int y1, sgx_2d_range_fn ok, + void *priv) +{ + unsigned long long lo, hi; + + if ((unsigned long long)(x1 + 1u) * s->bpp > s->stride) + return 0; + lo = s->va + (unsigned long long)y0 * s->stride; + hi = s->va + (unsigned long long)(y1 + 1u) * s->stride; + return ok(priv, lo, hi - lo); +} + +static inline int sgx_2d_refuse(struct sgx_2d_report *rep, unsigned int at, + const char *why, int code) +{ + if (rep) { + rep->at = at; + rep->why = why; + } + return code; +} + +/* + * Check a stream. Zero means acceptable, -1 malformed (a block cut short, a + * reserved bit set, a size of zero), 1 refused (a block the kernel does not + * let userspace emit, a surface outside the caller's bindings, a blit that + * reads a source that was never bound). + * + * Only blocks whose length is certain are parsed: DST_SURF and SRC_SURF (two + * dwords), SRC_OFF, FENCE and FLUSH (one), and BLIT with USE_PAT clear (four: + * command, fill colour, start, size). Every other block is refused, because a + * block whose length is misjudged puts the parser out of step with the engine + * - the fill colour of the next blit is then an address, which is exactly + * what work/twod-cursor/README.md saw fault at 0x8eadb000. CLIP_BH in + * particular does not take the two data words psb_reg.h suggests. + * + * base is the GPU address BIF_TWOD_REQ_BASE holds; a surface word carries + * an offset from it. The sequence-number page the kernel appends is not the + * stream's to name: the caller's range function refuses it. + */ +static inline int sgx_2d_check(const unsigned int *s, unsigned int n, + unsigned long long base, sgx_2d_range_fn ok, + void *priv, struct sgx_2d_report *rep) +{ + struct sgx_2d_surf_state dst = { 0, 0, 0, 0, 0 }; + struct sgx_2d_surf_state src = { 0, 0, 0, 0, 0 }; + unsigned int sx = 0, sy = 0; + int src_off = 0; + unsigned int i = 0; + + if (!s || !n || n > SGX_2D_MAX_DWORDS) + return sgx_2d_refuse(rep, 0, "empty or oversized stream", -1); + while (i < n) { + unsigned int w = s[i]; + + switch (w & SGX_2D_BH_MASK) { + case SGX_2D_DST_SURF_BH: + case SGX_2D_SRC_SURF_BH: { + struct sgx_2d_surf_state *st = + (w & SGX_2D_BH_MASK) == SGX_2D_DST_SURF_BH ? + &dst : &src; + unsigned int bpp = sgx_2d_format_bpp(w); + unsigned int stride = w & SGX_2D_SURF_STRIDE_MASK; + unsigned int addr; + + if (i + 2 > n) + return sgx_2d_refuse(rep, i, "surface block cut short", -1); + if (w & SGX_2D_SURF_RESERVED) + return sgx_2d_refuse(rep, i, "reserved surface bits", -1); + if (!bpp) + return sgx_2d_refuse(rep, i, "surface format", 1); + if (!stride || (stride & 3u) || stride < bpp) + return sgx_2d_refuse(rep, i, "surface stride", -1); + addr = s[i + 1]; + if (addr & ~SGX_2D_ADDR_MASK) + return sgx_2d_refuse(rep, i + 1, "surface address bits", -1); + st->fmt = w & SGX_2D_SURF_FORMAT_MASK; + st->bpp = bpp; + st->stride = stride; + st->va = base + addr; + st->set = 1; + if (st == &src) + src_off = 0; /* a new source wants its offset */ + i += 2; + break; + } + case SGX_2D_SRC_OFF_BH: + if (w & ~(SGX_2D_BH_MASK | SGX_2D_XY_MASK)) + return sgx_2d_refuse(rep, i, "reserved offset bits", -1); + sx = (w >> 12) & SGX_2D_COORD_MAX; + sy = w & SGX_2D_COORD_MAX; + src_off = 1; + i += 1; + break; + case SGX_2D_FENCE_BH: + case SGX_2D_FLUSH_BH: + /* psb_reg.h:290: the low 28 bits are ignored, so they + * are required clear rather than left to mean something + * on another revision. */ + if (w & ~SGX_2D_BH_MASK) + return sgx_2d_refuse(rep, i, "fence or flush data bits", -1); + i += 1; + break; + case SGX_2D_BLIT_BH: { + unsigned int rop = w & 0xffu; + unsigned int order = w & SGX_2D_COPYORDER_MASK; + unsigned int xy, wh, x0, y0, x1, y1; + + if (i + 4 > n) + return sgx_2d_refuse(rep, i, "blit block cut short", -1); + if (w & SGX_2D_BLIT_RESERVED) + return sgx_2d_refuse(rep, i, "reserved blit bits", -1); + /* Rotation swaps the source's extent; keying and + * blending need a CTRL block; clipping needs CLIP; a + * pattern needs PAT_SURF. None are let through, so the + * extent below is the whole of what the engine reads. */ + if (w & SGX_2D_ROT_MASK) + return sgx_2d_refuse(rep, i, "rotation", 1); + if (w & (SGX_2D_DSTCK_MASK | SGX_2D_SRCCK_MASK)) + return sgx_2d_refuse(rep, i, "colour key", 1); + if (w & (SGX_2D_CLIP_ENABLE | SGX_2D_ALPHA_ENABLE)) + return sgx_2d_refuse(rep, i, "clip or alpha enable", 1); + if (w & SGX_2D_USE_PAT) + return sgx_2d_refuse(rep, i, "pattern surface", 1); + if (((w >> SGX_2D_ROP3B_SHIFT) & 0xffu) != rop) + return sgx_2d_refuse(rep, i, "rop halves differ", 1); + xy = s[i + 2]; + wh = s[i + 3]; + if ((xy | wh) & ~SGX_2D_XY_MASK) + return sgx_2d_refuse(rep, i + 2, "reserved rectangle bits", -1); + if (!dst.set) + return sgx_2d_refuse(rep, i, "blit before a destination", 1); + if (!sgx_2d_rect(order, (xy >> 12) & SGX_2D_COORD_MAX, + xy & SGX_2D_COORD_MAX, + (wh >> 12) & SGX_2D_COORD_MAX, + wh & SGX_2D_COORD_MAX, + &x0, &y0, &x1, &y1)) + return sgx_2d_refuse(rep, i + 3, "rectangle", -1); + if (!sgx_2d_extent_ok(&dst, x1, y0, y1, ok, priv)) + return sgx_2d_refuse(rep, i, "destination outside the caller's bindings", 1); + if (sgx_2d_rop_reads_src(rop)) { + if (!src.set || !src_off) + return sgx_2d_refuse(rep, i, "source rop without a source and offset", 1); + if (!sgx_2d_rect(order, sx, sy, + (wh >> 12) & SGX_2D_COORD_MAX, + wh & SGX_2D_COORD_MAX, + &x0, &y0, &x1, &y1)) + return sgx_2d_refuse(rep, i, "source rectangle", -1); + if (!sgx_2d_extent_ok(&src, x1, y0, y1, ok, priv)) + return sgx_2d_refuse(rep, i, "source outside the caller's bindings", 1); + } + i += 4; + break; + } + default: + return sgx_2d_refuse(rep, i, "block type not accepted from userspace", 1); + } + } + return 0; +} + +/* The trailer the kernel appends: bind its own page as a 4-byte-stride + * 8888 surface and fill one pixel with the sequence number, then FENCE and + * FLUSH in the order pvr_copy() emits them. psb_blit_sequence() in the old + * driver's psb_sgx.c:198 wrote its fence the same way. */ +static inline void sgx_2d_trailer(unsigned int *w, unsigned int page_off, + unsigned int seq) +{ + w[0] = sgx_2d_surf_word(SGX_2D_DST_SURF_BH, SGX_2D_FMT_8888ARGB, 4u); + w[1] = page_off & SGX_2D_ADDR_MASK; + w[2] = SGX_2D_BLIT_BH | SGX_2D_ROT_NONE | SGX_2D_COPYORDER_TL2BR | + SGX_2D_USE_FILL | SGX_2D_ROP3(SGX_2D_ROP_PATCOPY); + w[3] = seq; + w[4] = SGX_2D_XY(0, 0); + w[5] = SGX_2D_XY(1, 1); + w[6] = SGX_2D_FENCE_BH; + w[7] = SGX_2D_FLUSH_BH; +} + +#endif /* _SGX_TWOD_H_ */ diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_uapi_drm.h mesa-26.2.2/src/gallium/drivers/sgx/sgx_uapi_drm.h --- mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_uapi_drm.h 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/sgx_uapi_drm.h 2026-09-08 10:57:36.685760277 +0200 @@ -0,0 +1,373 @@ +/* SPDX-License-Identifier: MIT */ +/* + * UAPI for a PowerVR SGX535 render node on Poulsbo, as a render side of the + * existing gma500 display driver. + * + * This is a proposal, not a shipped interface. It is written against what the + * hardware was measured to need - see MESA-DRI-GAPS.md and GL-IMPLEMENTATION.md + * - and deliberately against what the 2008 psb driver did, in three places: + * + * - No xhw round trip. psb pushed scene management into userspace and had + * the kernel ask it questions over a mailbox. The responder is open C now + * (tools/xpsb-open/xpsb_hw.c, ~900 lines), so it belongs in the kernel and + * userspace never sees a scene cookie. + * + * - No relocations. A frame currently applies 77 TA and 16 raster + * relocations; userspace assigns GPU addresses itself here and the kernel + * only validates that a named object is mapped where the stream says. + * Relocation-based submit is an argument upstream would not accept. + * + * - Real fences. dma_fence out of SUBMIT rather than the sequence numbers + * and psb_fence_wait of the old driver. + * + * Copyright (C) 2026 René Rebe + */ +#ifndef _UAPI_SGX_DRM_H_ +#define _UAPI_SGX_DRM_H_ + +/* One header, two contexts: the kernel has the real , the host + * tests have a stub. There used to be two copies of this file for that reason + * and they had already drifted - SGX_VM_SCENE existed in one and not the + * other. kernel/sgx_drm.h is a symlink to this now, so they cannot. */ +#ifdef __KERNEL__ +#include +#else +#include "drm.h" +#endif + +#if defined(__cplusplus) +extern "C" { +#endif + +#define DRM_SGX_GEM_NEW 0x00 +#define DRM_SGX_GEM_MAP 0x01 +#define DRM_SGX_GEM_WAIT 0x02 +#define DRM_SGX_VM_BIND 0x03 +#define DRM_SGX_SUBMIT 0x04 +#define DRM_SGX_GET_PARAM 0x05 +#define DRM_SGX_USE_BASE 0x06 +#define DRM_SGX_SCANOUT 0x07 +#define DRM_SGX_WAIT 0x08 +#define DRM_SGX_BLIT 0x09 +#define DRM_SGX_VISTEST 0x0a + +/* + * A buffer object. Nothing here is SGX specific except the flags: the display + * side of gma500 already has GEM, and a render node has to share it so a + * scanout buffer can be rendered into without a copy. + */ +struct drm_sgx_gem_new { + __u64 size; + __u32 flags; +#define SGX_BO_SCANOUT (1 << 0) /* must satisfy the display stride */ +#define SGX_BO_CPU_CACHED (1 << 1) /* the parameter heap is uncached */ + __u32 handle; /* out */ +}; + +struct drm_sgx_gem_map { + __u32 handle; + __u32 pad; + __u64 offset; /* out: mmap cookie */ +}; + +struct drm_sgx_gem_wait { + __u32 handle; + __u32 flags; + __s64 timeout_ns; +}; + +/* + * Explicit address binding. Userspace picks the GPU address, which is what + * lets the submit path be relocation free; the kernel programs the SGX MMU + * (gma500 already has mmu.c) and refuses overlaps. + * + * The address space is not flat: the SGX takes its parameter heap, its render + * targets and its PDS window through different requestors with different base + * registers, so a binding names which one it is for. + */ +struct drm_sgx_vm_bind { + __u32 handle; + __u32 flags; +#define SGX_BIND_UNBIND (1 << 0) + __u64 gpu_va; + __u64 offset; + __u64 size; + __u32 window; +#define SGX_VM_PARAM 0 /* the DPM parameter heap */ +#define SGX_VM_PDS 1 /* heap, USSE and PDS programs */ +#define SGX_VM_SURFACE 2 /* render targets, depth, textures */ +#define SGX_VM_RASTGEOM 3 +#define SGX_VM_SCENE 4 /* the scene and the DPM's own tables */ +#define SGX_VM_COUNT 5 /* windows, so nothing hardcodes four */ + __u32 pad; +}; + +/* + * One frame. + * + * The two register streams are what the hardware actually consumes: a TA + * stream that binds the vertex source and starts the tiler, and a raster + * stream that runs the 3D core over what the tiler binned. They are arrays of + * {register offset, value} pairs, the same shape psb_reg_submit() took, and + * the kernel validates every offset against a whitelist rather than trusting + * them - that check does not exist in the old driver and is the reason its + * submit path could not be upstreamed. + * + * There is no oom_cmds member. The out-of-memory recovery is the kernel's + * business now: it owns the scene, and work/heap-probe plus the recovery in + * xpsb_hw.c showed that getting it wrong loses geometry silently. Userspace + * cannot be trusted with a path whose failure mode is a slightly wrong frame. + */ +struct drm_sgx_submit { + __u64 ta_stream; /* __u32 pairs */ + __u64 raster_stream; + __u32 ta_stream_count; /* in dwords, so 2 per pair */ + __u32 raster_stream_count; + + /* The out-of-memory command stream. The fire path branches on whether + * it was given one - scene_fire_common() sets the cookie's bit 31 only + * when num_oom_cmds is non-zero - so this is not optional decoration + * for a frame that never runs out: without it the DPM does not recycle + * parameter pages and an animation walks off the end of the heap. */ + __u64 oom_stream; + __u32 oom_stream_count; + __u32 oom_pad; + + __u64 bo_handles; /* every object the frame touches */ + __u32 bo_count; + __u32 flags; +/* The most handles one submit may name. A sanity bound rather than a hardware + * one: the kernel copies the list with memdup_user() and nothing downstream is + * sized by it. It was 64, which put the driver's texture budget at 48 and cost + * ioquake3 a whole frame per swap when a scene sampled more than that. */ +#define SGX_SUBMIT_MAX_HANDLES 256u +#define SGX_SUBMIT_NO_PRESENT (1 << 0) +/* Ask for out_sync_fd. Without this no descriptor is created: the fire is + * synchronous and the fence is signalled before it is handed out, so a caller + * that does not wait on one was only leaking it. */ +#define SGX_SUBMIT_FENCE_OUT (1 << 1) +/* Come back once the render is fired rather than once it has ended, so the + * caller can build the next frame while this one is on the core. Nothing about + * the frame changes; what changes is who waits for it and when. + * + * The wait is not skipped, it is moved: the next submit drains it before it + * reprograms anything, and DRM_IOCTL_SGX_WAIT is where userspace drains it + * before it reads a surface back or overwrites state the render is still + * reading. A caller that sets this and then writes into its heap, its shader + * code or a bound texture without waiting is racing the core, and the + * corruption that follows is its own. */ +#define SGX_SUBMIT_DEFER_WAIT (1 << 2) +/* This frame has objects armed for the ISP's visibility counters, so the + * kernel has to harvest them when the render ends. The vendor's microkernel + * takes the same flag through the kick - SGXMKIF_RENDERFLAGS_GETVISRESULTS, + * "setting this flag will cause the uKernel to collect the visibility + * results" (services4/include/sgx_mkif_client.h:54-57), set from + * SGX_KICKTA_FLAGS_GETVISRESULTS in sgxkick_client.c:2338-2354 - and only + * then runs the accumulate-and-clear at 3d.asm:714-776. A frame without it + * arms no object, so the counters do not move and there is nothing to read. + */ +#define SGX_SUBMIT_VISTEST (1 << 3) + +/* Two samples a pixel in each axis - four in all, the only multisample mode + * this core has. SGX_FEATURE_MSAA_2X_IN_X and SGX_FEATURE_MSAA_2X_IN_Y are + * what would make a two-sample render target legal and the SGX535 block of + * the vendor's sgxfeaturedefs.h defines neither, so there is no flag for one. + * + * The kernel needs this because it, not userspace, builds the scene: the + * tail-pointer and region-header arrays hold one entry per sample tile, so a + * multisampled scene needs four times as many of each. A frame that programs + * EUR_CR_TE_AA without this flag would have the tiler write past the end of + * the region array. */ +#define SGX_SUBMIT_MSAA_2X2 (1 << 4) + + /* The render target's geometry, which the kernel needs to size the + * scene and the parameter heap - xpsb_scene_info() and + * dpm_param_pages(). Userspace does not get to pick the heap size: + * that estimate is what the recovery path depends on. */ + __u32 width, height; + + /* Must be -1. No path here waits on a fence before it fires, so a + * submit carrying one is refused rather than run out of the order the + * caller asked for. */ + __s32 in_sync_fd; + /* Out, and only when SGX_SUBMIT_FENCE_OUT asked for one; -1 otherwise. + */ + __s32 out_sync_fd; +}; + +/* + * USE base registers. + * + * A shader is fetched through one of thirteen base registers, and a PDS + * program names the register and an offset within it rather than an address. + * Those registers are a global hardware resource: a client that could write + * them could redirect another client's code fetch, so they are not in the + * submit whitelist and userspace cannot set them. + * + * Instead it asks. Given the address and size of some code and which data + * master will run it, the kernel allocates or reuses a base register and + * returns the register number and the offset within it - which is exactly what + * a PDS program needs to encode. The assignment lasts for the file. + */ +struct drm_sgx_use_base { + __u64 gpu_va; /* where the code is */ + __u32 size; + __u32 data_master; +/* The hardware's own encoding, which is not the obvious order: psb_regman.c + * defaults every managed register to PIXEL by writing 1 into the field. The + * kernel passes this value straight through to CR_USE_CODE_BASE, so naming + * PIXEL 0 here would have selected VERTEX on the device. */ +#define SGX_USE_DM_VERTEX 0 +#define SGX_USE_DM_PIXEL 1 + __u32 reg; /* out: 3..15 */ + __u32 offset; /* out: gpu_va - the register's base */ +}; + +/* + * The console framebuffer, as a render target. + * + * The 3D core renders into the memory the display is already scanning out, so + * a frame becomes visible without anything copying it. The kernel maps that + * memory into the caller's address space at the render-target window and says + * where and how big it is; there is no handle, because the buffer is the + * driver's and userspace never gets to own it. + */ +struct drm_sgx_scanout { + __u64 gpu_va; /* out: where it was mapped */ + __u32 pitch; /* out: bytes per row */ + __u32 width, height; /* out */ + __u32 size; /* out */ + __u32 pad; +}; + +/* + * Drain the device: return once the render the last submit fired has ended. + * + * Only meaningful with SGX_SUBMIT_DEFER_WAIT, and cheap when nothing is + * outstanding. There is no handle: the core runs one frame at a time, so + * "the pending frame" is unambiguous and waiting for it is waiting for all of + * it - the tiler, the render and the parameter-memory dealloc. + */ +struct drm_sgx_wait { + __u32 flags; + __u32 pad; +}; + +/* + * The ISP's visibility counters, which are what an occlusion query counts. + * + * Eight of them, EUR_CR_ISP_VISTEST_VISIBLE0..7 (sgx535defs.h:1378-1415), and + * on this core they are device registers and nothing else: the SGX535 feature + * block does not define SGX_FEATURE_VISTEST_IN_MEMORY (sgxfeaturedefs.h: + * 189-232), so there is no in-memory result the way later cores have. A + * submitted stream is register writes and userspace cannot read a register, so + * the counter has to be read on this side and handed back - which is why this + * is an ioctl of its own rather than a field somewhere. It cannot be a field + * in DRM_IOCTL_SGX_WAIT either: that one is DRM_IOW and giving it an output + * changes its command number, which would break every existing caller. + * + * What comes back is an accumulator per counter, not the register. The kernel + * reads all eight when a render armed with SGX_SUBMIT_VISTEST ends, adds them + * into these, and clears the registers - which is exactly what the vendor's + * microkernel does per render (3d.asm:714-776, the read-add-write into the + * client's eight-dword buffer followed by the all-eight CLEAR write at :767). + * Accumulating is not optional there and is not optional here: on SGX535 the + * client's SGX_KICKTA_FLAGS_DISABLE_ACCUMVISRESULTS is ignored and the flag is + * always set (sgxkick_client.c:2344-2351), and the accumulate runs "after + * every render including SPM partial renders" (3d.asm:1170-1191). A query that + * spans a frame split therefore still counts every part of itself. + * + * The accumulators are per open file and never reset. A caller reads them + * before it arms a query and again after, and the difference is that query's + * count - which is also what lets several queries share one file without a + * reset racing between them. + */ +#define SGX_VISTEST_REGS 8 + +struct drm_sgx_vistest { + __u32 flags; +/* Drain a deferred render before reading, so the counts include the frame the + * last submit fired. Without it the read is a snapshot of what has landed. */ +#define SGX_VISTEST_WAIT (1 << 0) + /* Out: this file still has a render outstanding, so a count that has + * not landed yet is missing from what follows. Always zero when + * SGX_VISTEST_WAIT was asked for and the drain succeeded. */ + __u32 pending; + __u64 count[SGX_VISTEST_REGS]; /* out */ +}; + +/* + * A 2D job: copies and solid fills on the SGX535's 2D block, which is a + * separate engine from the 3D pipeline - no scene, no shaders, no tiler. + * + * The stream is the engine's own command format, block headers as __u32 + * dwords (driver/gma500/sgx_twod.h), and the kernel checks every block before + * it writes one to the slave port: only surfaces, a source offset, blits, + * fences and flushes are accepted, and every surface address - an offset from + * BIF_TWOD_REQ_BASE, which is the surface window's start - has to fall inside + * a binding this file made. A stream is not a register write: nothing in it + * can reach a register, so the whitelist in sgx_check.h does not grow. + * + * Synchronous. The engine runs behind the same lock as the 3D core and shares + * its page-directory register, so a blit drains any render still on the core + * first and has finished when the ioctl returns; the next submit therefore + * sees what it wrote, and it saw what the render before it wrote. A fence is + * handed out already signalled, like the submit's. + */ +struct drm_sgx_blit { + __u64 stream; /* __u32 dwords */ + __u32 stream_count; /* in dwords */ + __u32 flags; +#define SGX_BLIT_FENCE_OUT (1 << 0) +/* Come back once the stream is on the engine rather than once the engine has + * run it, so the caller can carry on while it works. The wait is not skipped, + * it is moved: the kernel drains it wherever it drains a deferred render - at + * the head of the next submit, in VM_BIND before an address can change under + * the engine, in GEM_WAIT and in DRM_IOCTL_SGX_WAIT, which is where userspace + * drains it before reading a surface. A caller that sets this and then reads + * or rebinds without waiting is racing the engine. Refused together with + * SGX_BLIT_FENCE_OUT: that fence is handed out signalled. */ +#define SGX_BLIT_DEFER_WAIT (1 << 1) + __u64 bo_handles; /* every object the stream touches */ + __u32 bo_count; + __u32 pad; + __s32 in_sync_fd; /* must be -1 */ + __s32 out_sync_fd; /* out, with SGX_BLIT_FENCE_OUT */ +}; + +/* + * Parameters. Kept small on purpose: anything userspace can derive, it should. + */ +struct drm_sgx_get_param { + __u32 param; +/* EUR_CR_CORE_REVISION as the hardware reports it: designer in bits 31:24, + * major 23:16, minor 15:8, maintenance 7:0. SGX535 rev 1.2.1 is 0x00010201. */ +#define SGX_PARAM_CORE_ID 0 +#define SGX_PARAM_CORE_CLOCK 1 /* 200 MHz, from the message bus */ +#define SGX_PARAM_VM_START 2 /* per-window, index in value */ +#define SGX_PARAM_VM_SIZE 3 +/* The GPU address BIF_TWOD_REQ_BASE holds; a 2D surface word carries an + * offset from it, 28 bits wide. Refused by a kernel without the 2D path, + * which is how userspace learns it has none. */ +#define SGX_PARAM_TWOD_BASE 4 + __u32 index; + __u64 value; /* in for index, out for value */ +}; + +#define DRM_IOCTL_SGX_GEM_NEW DRM_IOWR(DRM_COMMAND_BASE + DRM_SGX_GEM_NEW, struct drm_sgx_gem_new) +#define DRM_IOCTL_SGX_GEM_MAP DRM_IOWR(DRM_COMMAND_BASE + DRM_SGX_GEM_MAP, struct drm_sgx_gem_map) +#define DRM_IOCTL_SGX_GEM_WAIT DRM_IOW (DRM_COMMAND_BASE + DRM_SGX_GEM_WAIT, struct drm_sgx_gem_wait) +#define DRM_IOCTL_SGX_VM_BIND DRM_IOW (DRM_COMMAND_BASE + DRM_SGX_VM_BIND, struct drm_sgx_vm_bind) +#define DRM_IOCTL_SGX_SUBMIT DRM_IOWR(DRM_COMMAND_BASE + DRM_SGX_SUBMIT, struct drm_sgx_submit) +#define DRM_IOCTL_SGX_GET_PARAM DRM_IOWR(DRM_COMMAND_BASE + DRM_SGX_GET_PARAM, struct drm_sgx_get_param) +#define DRM_IOCTL_SGX_USE_BASE DRM_IOWR(DRM_COMMAND_BASE + DRM_SGX_USE_BASE, struct drm_sgx_use_base) +#define DRM_IOCTL_SGX_SCANOUT DRM_IOWR(DRM_COMMAND_BASE + DRM_SGX_SCANOUT, struct drm_sgx_scanout) +#define DRM_IOCTL_SGX_WAIT DRM_IOW (DRM_COMMAND_BASE + DRM_SGX_WAIT, struct drm_sgx_wait) +#define DRM_IOCTL_SGX_BLIT DRM_IOWR(DRM_COMMAND_BASE + DRM_SGX_BLIT, struct drm_sgx_blit) +#define DRM_IOCTL_SGX_VISTEST DRM_IOWR(DRM_COMMAND_BASE + DRM_SGX_VISTEST, struct drm_sgx_vistest) + +#if defined(__cplusplus) +} +#endif + +#endif /* _UAPI_SGX_DRM_H_ */ diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_winsys.c mesa-26.2.2/src/gallium/drivers/sgx/sgx_winsys.c --- mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_winsys.c 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/sgx_winsys.c 2026-09-08 10:57:36.681264140 +0200 @@ -0,0 +1,1598 @@ +/* SPDX-License-Identifier: MIT */ +/* + * Copyright (C) 2026 René Rebe + */ +#include "sgx_winsys.h" +#include "sgx_drm.h" +#include "sgx_twod.h" + +#include +#include +#include +#include +#include + +/* Mirrors the table the kernel enforces. Keeping a copy here rather than + * asking every time is only safe because GET_PARAM reports the same values and + * sgx_winsys_create() checks them - a driver that assumed would drift. */ +/* dyn is where bump allocation starts. It is the window start for the windows + * nothing binds at a fixed address, and above the reserved block for the two + * that the frame does: the surface window, where the render target, the depth + * buffer and the two texture units sit at fixed addresses, and the raster + * geometry window, where the frame's own vertex and index block does. + * + * Running the bump allocator from the window start in either of those hands a + * caller's buffer the address the frame is about to want. It is not a + * theoretical collision: a display server allocates its vertex buffer before + * it sets a framebuffer, so the frame's raster geometry block was refused and + * nothing rendered at all. sgx_ctx_va and sgx_ctx_size in the Gallium context + * are the other half of this and have to agree with it. */ +static struct { uint64_t start, size, next, dyn; } win[SGX_VM_COUNT] = { + { 0x31000000ull, 0x0ee00000ull, 0, 0x31000000ull }, + { 0x20010000ull, 0x00ff0000ull, 0, 0x20010000ull }, + { 0x80000000ull, 0x20000000ull, 0, 0x85000000ull }, + { 0x30000000ull, 0x01000000ull, 0, 0x30020000ull }, + { 0x40000000ull, 0x02000000ull, 0, 0x40000000ull }, +}; + +/* Live bindings, sorted by address, one list per window. A bump pointer was + * enough while a context allocated once and kept everything, but anything that + * allocates and frees - a display server, or a test that walks texture sizes - + * exhausts the window instead of reusing what it returned. */ +struct sgx_vm_range { + uint64_t va, size; + struct sgx_vm_range *next; +}; + +/* The USE base registers this file may be granted, and how much of the address + * space one covers. Both are the kernel's, in sgx_use.h: thirteen registers + * from three, each covering 512 KiB. Mirrored rather than included because the + * winsys builds against the uapi header alone. */ +#define SGX_USE_CACHE_NUM 13u +#define SGX_USE_CACHE_WINDOW 0x80000u + +/* Released buffers kept for reuse, and the most memory they may hold between + * them. Both are policy rather than hardware: enough to cover a display + * server's pixmap churn without hoarding. */ +/* Slots, bounded in bytes by SGX_BO_CACHE_BYTES below rather than by this. + * At sixty-four the slots filled with large one-off objects and then every + * streaming buffer a frame freed was closed instead of parked - a client that + * recycles two sizes missed on every allocation while megabytes sat parked, + * which for ioquake3 was thirty-eight fresh GEM objects a frame for the kernel + * to zero and set the cache attributes of. An entry is forty bytes. */ +#define SGX_BO_CACHE_NUM 512u +#define SGX_BO_CACHE_BYTES (24u * 1024u * 1024u) + +/* getenv() walks the environment, and the debug test below sits on a path + * taken per relocation. Asked once. */ +/* Asked once, like the debug test: this sits on the free path. */ +static int sgx_ws_env(const char *name, signed char *cache) +{ + if (*cache < 0) + *cache = getenv(name) != NULL; + return *cache; +} + +#define SGX_ENVS_WS(n) __extension__({ \ + static signed char sgx_ws_env_cached_ = -1; \ + sgx_ws_env(n, &sgx_ws_env_cached_); \ +}) + +/* Where the winsys writes: SGX_LOG's file when set, since a display server + * closes stderr; the same sink the driver's sgx_log() uses. */ +static FILE *sgx_ws_log(void) +{ + static FILE *f; + static int tried; + + if (!tried) { + const char *e = getenv("SGX_LOG"); + + tried = 1; + if (e && *e) { + f = fopen(e, "ae"); + if (f) + setvbuf(f, NULL, _IOLBF, 0); + } + } + return f ? f : stderr; +} + +static int sgx_ws_debug(void) +{ + static int on = -1; + + if (on < 0) { + const char *e = getenv("SGX_DEBUG"); + + on = e && *e && *e != '0'; + } + return on; +} + +/* One entry per place a wait came from, allocated as they turn up - the set is + * whatever the caller happens to have, not a table to be kept in step. The + * name is __func__, so it is a pointer comparison. */ +struct sgx_wait_site { + const char *where; + unsigned n; + struct sgx_wait_site *next; +}; + +struct sgx_winsys { + struct sgx_ioctl_ops ops; + struct sgx_vm_range *live[SGX_VM_COUNT]; + uint64_t next[SGX_VM_COUNT]; + int fd; /* -1 unless a real device backs this */ + /* Whether a submit may come back before the render has ended, and + * whether one has. The pair is here rather than in the Gallium context + * because the frame is not the only thing that reads GPU memory: every + * resource map goes through this layer, and that is the choke point + * where the wait has to happen. */ + int async, busy; + /* The next submit carries SGX_SUBMIT_VISTEST, because the frame it + * builds has an object armed for a visibility counter. Consumed by + * that submit: arming is per frame, not per context. */ + int vistest_next; + /* Samples along one axis, 1 or 2, for the next submit. Here rather + * than a submit argument because the kernel needs it to size the + * scene, and every caller of sgx_submit() would otherwise have to + * carry a parameter that is only ever the framebuffer's. */ + unsigned msaa_axis; + /* A wait that failed means a frame the core never finished. It is + * sticky because that is what GL asks about: a context wants to know + * that it lost work at some point, not whether the last wait failed. */ + int lost; + /* Bumped per loss, so a caller can tell a new one from the last. */ + unsigned lost_gen; + unsigned deferred; /* submits that did not wait */ + struct sgx_wait_site *sites; /* where the waits landed */ + /* The USE base registers the kernel has handed out, mirrored so a + * repeat of the same question is answered here. sgx_use_grab() is a + * search for a register that already covers the range and only claims + * a free one when none does, and nothing releases them for the life + * of the file - so this can be an exact copy rather than a guess. + * + * It matters because the question is asked per relocation while a + * frame is built: one twm menu asked it 2844 times, which was 93% of + * the server's system time and most of the second the menu took to + * appear. */ + struct { + uint32_t base; /* window base, va - offset */ + uint32_t reg; + uint8_t dm; + uint8_t valid; + } use[SGX_USE_CACHE_NUM]; + unsigned use_hit, use_miss; + /* Buffers a caller has released, kept rather than handed back to the + * kernel. Creating one costs set_pages_array_wc() over its pages, a + * zeroing memset and the page allocation itself, and freeing costs + * set_pages_array_wb() - which is why a twm menu, which creates and + * destroys forty of them, spent a quarter of its time in + * change_page_attr_set_clr() and memset() with the GPU idle. + * + * A parked buffer keeps its mapping and its binding, so a hit also + * saves the map and the ten binds each one costs. + * + * safe_after is what makes reuse safe: a render already fired may + * still be reading the buffer, so an entry parked while the core was + * busy cannot be handed out until a wait has completed. */ + struct { + struct sgx_bo bo; + unsigned safe_after; + uint8_t valid; + } cache[SGX_BO_CACHE_NUM]; + unsigned completed; /* waits that have finished */ + uint64_t cached_bytes; + unsigned bo_hit, bo_miss; + /* The 2D block: whether the kernel has it, and where its request + * base is. -1 until asked. */ + int twod; + uint64_t twod_base; + unsigned twod_jobs, twod_blits; + /* The blits promised and not yet handed to the kernel. + * + * A job costs one synchronous ioctl - about 48 us on the part - and a + * display server's copies come a glyph or a rectangle at a time, so + * one job per copy made the engine 25x slower than the CPU at text + * while being 12x quicker per byte. The job stays open instead and + * the blits accumulate in it; what closes it is something that has to + * observe the result, which is what the flush hooks below are. */ + struct sgx_twod *job; + unsigned twod_queued, twod_flushes; + /* A job handed over with SGX_BLIT_DEFER_WAIT and not yet waited for. + * Kept apart from busy, which is the render's: the kernel drains them + * at the same places but only a render advances the completion count + * that the per-buffer busy test is written against. */ + int twod_busy; +}; + +/* Close the open job. The addresses in a job are absolute and were taken when + * the blit was appended, so anything that can move an object, free it, read it + * or render over it has to come after the engine has run - these are what put + * it there. twod_flush_named() closes the job only if it names that buffer. */ +static int twod_flush_all(struct sgx_winsys *ws, const char *why); +static void twod_flush_named(struct sgx_winsys *ws, const struct sgx_bo *bo, + const char *why); +static void range_drop(struct sgx_winsys *ws, uint32_t window, uint64_t va); +static int range_add(struct sgx_winsys *ws, uint32_t window, uint64_t va, + uint64_t size); +static void wait_count(struct sgx_winsys *ws, const char *where); + +int sgx_winsys_fd(struct sgx_winsys *ws) +{ + return ws ? ws->fd : -1; +} + +void sgx_winsys_set_fd(struct sgx_winsys *ws, int fd) +{ + if (ws) + ws->fd = fd; +} + +/* Deferred waiting is opt-in, and only on a real device: a test's winsys has + * no kernel behind it to drain, and its mock would be asked for an ioctl it + * does not implement. + * + * And only on a kernel that has the drain, which is asked rather than assumed: + * an older module refuses the ioctl, and it would also refuse the submit flag - + * so without this probe a driver built against a new header and run on an old + * module would lose every frame instead of a little overlap. */ +/* The sample count the frames that follow are built for. Refuses anything the + * part cannot tile rather than rounding it, and reports what is in force. */ +int sgx_winsys_set_msaa(struct sgx_winsys *ws, unsigned samples) +{ + /* One sample a pixel or 2x2; the part reaches no other count - + * xpsb_msaa_axis() is the same rule with the citations. */ + unsigned axis = samples == 1u ? 1u : samples == 4u ? 2u : 0u; + + if (!ws || !axis) + return -EINVAL; + ws->msaa_axis = axis; + return 0; +} + +unsigned sgx_winsys_msaa(const struct sgx_winsys *ws) +{ + if (!ws || !ws->msaa_axis) + return 1; + return ws->msaa_axis * ws->msaa_axis; +} + +void sgx_winsys_set_async(struct sgx_winsys *ws, int on) +{ + struct drm_sgx_wait a; + + if (!ws || ws->fd < 0) + return; + ws->async = 0; + if (!on) + return; + memset(&a, 0, sizeof a); + if (ws->ops.ioctl(ws->ops.dev, DRM_IOCTL_SGX_WAIT, &a)) { + if (getenv("SGX_DEBUG")) + fprintf(sgx_ws_log(), "sgx: the kernel has no drain; the " + "submit stays synchronous\n"); + return; + } + ws->async = 1; +} + +static void wait_count(struct sgx_winsys *ws, const char *where) +{ + struct sgx_wait_site *s; + + for (s = ws->sites; s; s = s->next) + if (s->where == where) { + s->n++; + return; + } + s = calloc(1, sizeof *s); + if (!s) + return; + s->where = where; + s->n = 1; + s->next = ws->sites; + ws->sites = s; +} + +/* Wait for the render the last submit fired, if the ioctl came back before it + * had ended. Everything that reads or overwrites memory the GPU may still be + * reading calls this first; it costs a branch when nothing is outstanding. */ +int sgx_wait_idle_where(struct sgx_winsys *ws, const char *where) +{ + struct drm_sgx_wait a; + int ret; + + if (!ws) + return 0; + /* Before the test below, not after it: a caller that waits is about + * to look at memory, and a promised blit is as invisible as an + * unfinished render. This is the hook every CPU map goes through. */ + twod_flush_all(ws, where ? where : "wait"); + if (!ws->busy && !ws->twod_busy) + return 0; + /* Which site actually blocks a frame, counted per caller: a wait that + * never fires and a wait that fires every frame look the same from a + * frame rate. SGX_WAIT_STATS prints the tally at screen teardown. */ + if (SGX_ENVS_WS("SGX_WAIT_STATS")) + fprintf(sgx_ws_log(), "sgx: wait at %s\n", + where ? where : "wait"); + if (getenv("SGX_ASYNC_STATS")) + wait_count(ws, where ? where : "?"); + memset(&a, 0, sizeof a); + ret = ws->ops.ioctl(ws->ops.dev, DRM_IOCTL_SGX_WAIT, &a); + /* Cleared either way: a failed wait is a frame that will not arrive, + * and the kernel has already dropped it. Leaving the flag set would + * make every later map retry a wait for nothing. + * + * The drain covers a deferred 2D job as well, but only a render + * advances the completion count: that count is what the per-buffer + * busy test compares against, and it counts renders. */ + ws->twod_busy = 0; + if (!ws->busy) + return ret; + ws->busy = 0; + ws->completed++; + if (ret) { + /* Unconditionally: this is a frame that will not arrive, and + * behind SGX_DEBUG a locked core looked like a slow one. */ + ws->lost = 1; + ws->lost_gen++; + fprintf(sgx_ws_log(), "sgx: the render did not finish: %d - the " + "frame is lost\n", ret); + } + return ret; +} + +uint64_t sgx_window_start(uint32_t w) +{ + return w < SGX_VM_COUNT ? win[w].start : 0; +} + +uint64_t sgx_window_size(uint32_t w) +{ + return w < SGX_VM_COUNT ? win[w].size : 0; +} + +struct sgx_winsys *sgx_winsys_create(const struct sgx_ioctl_ops *ops) +{ + struct sgx_winsys *ws; + unsigned i; + + if (!ops || !ops->ioctl) + return NULL; + ws = calloc(1, sizeof *ws); + if (!ws) + return NULL; + ws->ops = *ops; + ws->fd = -1; + + /* The kernel is the authority on the address map; check rather than + * assume the compiled-in copy still matches it. */ + for (i = 0; i < SGX_VM_COUNT; i++) { + struct drm_sgx_get_param p; + + memset(&p, 0, sizeof p); + p.param = SGX_PARAM_VM_START; p.index = i; + if (ws->ops.ioctl(ws->ops.dev, DRM_IOCTL_SGX_GET_PARAM, &p) || + p.value != win[i].start) + goto bad; + memset(&p, 0, sizeof p); + p.param = SGX_PARAM_VM_SIZE; p.index = i; + if (ws->ops.ioctl(ws->ops.dev, DRM_IOCTL_SGX_GET_PARAM, &p) || + p.value != win[i].size) + goto bad; + /* Bump allocation starts past whatever the frame reserves in + * that window, not at the window's start - starting at the + * start is what handed a caller's buffer the address the + * frame's own block needs. */ + ws->next[i] = win[i].dyn; + } + ws->twod = -1; + return ws; +bad: + free(ws); + return NULL; +} + +void sgx_winsys_destroy(struct sgx_winsys *ws) +{ + unsigned w; + + if (!ws) + return; + twod_flush_all(ws, "destroy"); + sgx_twod_abort(ws->job); + ws->job = NULL; + if (getenv("SGX_ASYNC_STATS") && ws->deferred) { + const struct sgx_wait_site *s; + + fprintf(sgx_ws_log(), "sgx: %u deferred submit(s); the waits landed " + "at:\n", ws->deferred); + for (s = ws->sites; s; s = s->next) + fprintf(sgx_ws_log(), "sgx: %-28s %u\n", s->where, s->n); + } + while (ws->sites) { + struct sgx_wait_site *d = ws->sites; + + ws->sites = d->next; + free(d); + } + sgx_bo_cache_drain(ws); + for (w = 0; w < SGX_VM_COUNT; w++) + while (ws->live[w]) { + struct sgx_vm_range *d = ws->live[w]; + + ws->live[w] = d->next; + free(d); + } + free(ws); +} + +int sgx_bo_new(struct sgx_winsys *ws, struct sgx_bo *bo, uint64_t size, + uint32_t flags) +{ + struct drm_sgx_gem_new a; + int ret; + + if (!ws || !bo || !size) + return -EINVAL; + a.size = (size + 0xfff) & ~0xfffull; + /* A buffer of exactly this size that nothing can still be reading. + * Exactly, not at least: the caller is told the size it got and the + * frame is built from it, so handing back a larger one would describe + * a surface that is not the one asked for. */ + if (!flags) { + unsigned i; + + for (i = 0; i < SGX_BO_CACHE_NUM; i++) + if (ws->cache[i].valid && + ws->cache[i].bo.size == a.size && + ws->completed >= ws->cache[i].safe_after) { + *bo = ws->cache[i].bo; + ws->cache[i].valid = 0; + ws->cached_bytes -= bo->size; + ws->bo_hit++; + return 0; + } + } + ws->bo_miss++; + /* SGX_BO_CACHE_STAT says how the cache is doing. A miss is a fresh + * GEM object: the kernel zeroes its pages and sets their cache + * attributes, which is the change_page_attr_set_clr and memset this + * cache was written to avoid. */ + if (SGX_ENVS_WS("SGX_BO_CACHE_STAT")) + fprintf(stderr, "sgx: bo cache: %u hit, %u miss, %llu KiB " + "parked, this miss %llu KiB\n", ws->bo_hit, + ws->bo_miss, + (unsigned long long)(ws->cached_bytes >> 10), + (unsigned long long)(a.size >> 10)); + memset(&a, 0, sizeof a); + a.size = (size + 0xfff) & ~0xfffull; + a.flags = flags; + ret = ws->ops.ioctl(ws->ops.dev, DRM_IOCTL_SGX_GEM_NEW, &a); + if (ret) + return ret; + memset(bo, 0, sizeof *bo); + bo->handle = a.handle; + bo->size = a.size; + return 0; +} + +int sgx_bo_map(struct sgx_winsys *ws, struct sgx_bo *bo) +{ + struct drm_sgx_gem_map a; + int ret; + + if (!ws || !bo || !ws->ops.mmap) + return -EINVAL; + memset(&a, 0, sizeof a); + a.handle = bo->handle; + ret = ws->ops.ioctl(ws->ops.dev, DRM_IOCTL_SGX_GEM_MAP, &a); + if (ret) + return ret; + bo->map = ws->ops.mmap(ws->ops.dev, a.offset, (size_t)bo->size); + return bo->map ? 0 : -ENOMEM; +} + +/* The header's count and the UAPI's are the same eight registers. */ +_Static_assert(SGX_VISTEST_COUNTERS == SGX_VISTEST_REGS, + "the winsys and the UAPI disagree on the counter count"); + +void sgx_vistest_arm(struct sgx_winsys *ws) +{ + if (ws) + ws->vistest_next = 1; +} + +int sgx_vistest_read(struct sgx_winsys *ws, int wait, uint64_t *counts, + int *pending) +{ + struct drm_sgx_vistest a; + unsigned i; + int ret; + + if (!ws || !counts) + return -EINVAL; + /* A promised blit is not a render and adds nothing here, but the wait + * this asks for is the same drain every other reader goes through - + * so it goes through the same door, which flushes the open job. */ + if (wait) + (void)sgx_wait_idle(ws); + memset(&a, 0, sizeof a); + a.flags = wait ? SGX_VISTEST_WAIT : 0u; + ret = ws->ops.ioctl(ws->ops.dev, DRM_IOCTL_SGX_VISTEST, &a); + if (ret) + return ret; + for (i = 0; i < SGX_VISTEST_COUNTERS; i++) + counts[i] = a.count[i]; + if (pending) + *pending = a.pending != 0; + return 0; +} + +int sgx_get_scanout(struct sgx_winsys *ws, struct sgx_scanout *out) +{ + struct drm_sgx_scanout a; + int ret; + + if (!ws || !out) + return -EINVAL; + memset(&a, 0, sizeof a); + ret = ws->ops.ioctl(ws->ops.dev, DRM_IOCTL_SGX_SCANOUT, &a); + if (ret) + return ret; + out->gpu_va = a.gpu_va; + out->pitch = a.pitch; + out->width = a.width; + out->height = a.height; + out->size = a.size; + return 0; +} + +/* The kernel's own test, in sgx_use_grab(): the range has to sit inside the + * register's 512 KiB window and want the same data master. */ +static int use_covers(uint32_t base, uint64_t va, uint32_t size) +{ + return base <= va && va - base < SGX_USE_CACHE_WINDOW && + size < SGX_USE_CACHE_WINDOW - (uint32_t)(va - base); +} + +unsigned sgx_winsys_completed(const struct sgx_winsys *ws) +{ + return ws ? ws->completed : 0u; +} + +/* Named by the frame being submitted: it is idle again once one more wait has + * completed. */ +void sgx_bo_mark_busy(struct sgx_winsys *ws, struct sgx_bo *bo) +{ + if (ws && bo) + bo->busy_gen = ws->completed + 1u; +} + +int sgx_bo_is_busy(const struct sgx_winsys *ws, const struct sgx_bo *bo) +{ + return ws && bo && ws->busy && bo->busy_gen > ws->completed; +} + +int sgx_use_base(struct sgx_winsys *ws, uint64_t gpu_va, uint32_t size, + uint32_t data_master, uint32_t *reg, uint32_t *offset) +{ + struct drm_sgx_use_base a; + unsigned i; + int ret; + + if (!ws || !reg || !offset) + return -EINVAL; + for (i = 0; i < SGX_USE_CACHE_NUM; i++) + if (ws->use[i].valid && ws->use[i].dm == data_master && + use_covers(ws->use[i].base, gpu_va, size)) { + *reg = ws->use[i].reg; + *offset = (uint32_t)(gpu_va - ws->use[i].base); + ws->use_hit++; + return 0; + } + ws->use_miss++; + memset(&a, 0, sizeof a); + a.gpu_va = gpu_va; + a.size = size; + a.data_master = data_master; + ret = ws->ops.ioctl(ws->ops.dev, DRM_IOCTL_SGX_USE_BASE, &a); + if (ret) + return ret; + if (sgx_ws_debug()) + fprintf(sgx_ws_log(), "sgx: use base: va 0x%llx size %u dm %u -> " + "reg %u offset 0x%x\n", (unsigned long long)gpu_va, + size, data_master, a.reg, a.offset); + /* Record the window the kernel granted. Its base is what the address + * was less the offset it came back with, which is the same value the + * kernel keeps. */ + for (i = 0; i < SGX_USE_CACHE_NUM; i++) + if (!ws->use[i].valid) { + ws->use[i].base = (uint32_t)(gpu_va - a.offset); + ws->use[i].reg = a.reg; + ws->use[i].dm = (uint8_t)data_master; + ws->use[i].valid = 1; + break; + } + *reg = a.reg; + *offset = a.offset; + return 0; +} + +static int bind_at(struct sgx_winsys *ws, struct sgx_bo *bo, uint32_t window, + uint64_t va) +{ + struct drm_sgx_vm_bind a; + int ret; + + if (va < win[window].start || + va + bo->size > win[window].start + win[window].size) + return -ENOSPC; + /* The kernel decides whether an address is free, and it is the only + * one that can: this layer's list of live ranges is a mirror, and a + * mirror that has drifted refuses a bind the kernel would have taken. + * There used to be a check here that did exactly that. It was added + * to give a nicer error for a page the kernel reserved inside this + * window, which the 2D block no longer has - and what it cost was + * that any disagreement between the two became a render target that + * did not get bound at all, which the part reports as a fault inside + * the target's own address. */ + + /* The job's stream carries this object at the address it has now. */ + twod_flush_named(ws, bo, "bind"); + + memset(&a, 0, sizeof a); + a.handle = bo->handle; + a.gpu_va = va; + a.size = bo->size; + a.window = window; + ret = ws->ops.ioctl(ws->ops.dev, DRM_IOCTL_SGX_VM_BIND, &a); + if (getenv("SGX_DEBUG")) + fprintf(sgx_ws_log(), "sgx: bind fd %d win %u handle %u at 0x%llx " + "+0x%llx -> %d\n", ws->fd, window, bo->handle, + (unsigned long long)va, (unsigned long long)bo->size, + ret); + if (ret) + return ret; + /* Every binding is recorded, the ones at a caller-chosen address + * included: the frame's fixed slots sit in the same windows the + * allocator hands out from, and an unrecorded range is one it will + * hand out twice. */ + if (range_add(ws, window, va, bo->size)) { + a.flags = SGX_BIND_UNBIND; + (void)ws->ops.ioctl(ws->ops.dev, DRM_IOCTL_SGX_VM_BIND, &a); + return -ENOMEM; + } + bo->gpu_va = va; + bo->window = window; + return 0; +} + +uint64_t sgx_window_bytes(uint32_t window) +{ + return window < SGX_VM_COUNT ? win[window].size : 0; +} + +int sgx_winsys_reserve(struct sgx_winsys *ws, uint32_t window, uint64_t upto) +{ + if (!ws || window >= SGX_VM_COUNT) + return -EINVAL; + if (upto < win[window].start || + upto > win[window].start + win[window].size) + return -EINVAL; + if (upto > ws->next[window]) + ws->next[window] = upto; + return 0; +} + +int sgx_bo_unbind(struct sgx_winsys *ws, struct sgx_bo *bo) +{ + struct drm_sgx_vm_bind a; + int ret; + + if (!ws || !bo) + return -EINVAL; + if (!bo->gpu_va) + return 0; + twod_flush_named(ws, bo, "unbind"); + memset(&a, 0, sizeof a); + a.handle = bo->handle; + a.flags = SGX_BIND_UNBIND; + a.gpu_va = bo->gpu_va; + a.size = bo->size; + a.window = bo->window; + ret = ws->ops.ioctl(ws->ops.dev, DRM_IOCTL_SGX_VM_BIND, &a); + if (ret) + return ret; + range_drop(ws, bo->window, bo->gpu_va); + bo->gpu_va = 0; + return 0; +} + +int sgx_bo_bind_at(struct sgx_winsys *ws, struct sgx_bo *bo, uint32_t window, + uint64_t gpu_va) +{ + if (!ws || !bo || window >= SGX_VM_COUNT) + return -EINVAL; + if (gpu_va & 0xfff) + return -EINVAL; + return bind_at(ws, bo, window, gpu_va); +} + +/* Record a binding so the address is not handed out twice, and so it can be + * reused once it comes back. A range that cannot be recorded is not bound: a + * silent omission here is a double allocation later. */ +static int range_add(struct sgx_winsys *ws, uint32_t window, uint64_t va, + uint64_t size) +{ + struct sgx_vm_range *r, **pp; + + r = calloc(1, sizeof *r); + if (!r) + return -ENOMEM; + r->va = va; + r->size = size; + for (pp = &ws->live[window]; *pp && (*pp)->va < va; pp = &(*pp)->next) + ; + r->next = *pp; + *pp = r; + return 0; +} + +static void range_drop(struct sgx_winsys *ws, uint32_t window, uint64_t va) +{ + struct sgx_vm_range **pp; + + for (pp = &ws->live[window]; *pp; pp = &(*pp)->next) + if ((*pp)->va == va) { + struct sgx_vm_range *d = *pp; + + *pp = d->next; + free(d); + return; + } +} + +/* First fit above the window's dynamic start, in the gaps between what is + * live. */ +/* How far past a refused address to look next, and how many times. The + * kernel's own reservations are megabytes wide, so a page at a time would + * cross the scenes in two thousand ioctls; the two together bound the search + * at a whole window. */ +#define SGX_BIND_STEP 0x10000ull +#define SGX_BIND_MAX_TRIES 512u + +static uint64_t range_find(struct sgx_winsys *ws, uint32_t window, + uint64_t size, uint64_t above) +{ + uint64_t va = (win[window].dyn + 0xfff) & ~0xfffull; + + if (above > va) + va = (above + 0xfff) & ~0xfffull; + uint64_t end = win[window].start + win[window].size; + const struct sgx_vm_range *r; + + for (r = ws->live[window]; r; r = r->next) { + if (r->va + r->size <= va) + continue; + if (va + size <= r->va) + break; /* fits in the gap before r */ + va = (r->va + r->size + 0xfff) & ~0xfffull; + } + return va + size <= end ? va : 0; +} + +int sgx_bo_bind(struct sgx_winsys *ws, struct sgx_bo *bo, uint32_t window) +{ + uint64_t va; + int ret; + + if (!ws || !bo || window >= SGX_VM_COUNT) + return -EINVAL; + /* The gap list is this file's own bindings, and the kernel reserves + * address space nothing here ever asked for: the scenes, the DPM's + * page table and the out-of-memory table all sit in the scene window. + * So a gap that looks free can still come back EEXIST - and a refused + * bind is a context that draws nothing at all, which is worth more + * than one ioctl to avoid. Step over the reservation and take the next + * gap instead of giving up. */ + { + unsigned tries; + + for (tries = 0, va = range_find(ws, window, bo->size, 0); + va && tries < SGX_BIND_MAX_TRIES; + tries++, va = range_find(ws, window, bo->size, + va + SGX_BIND_STEP)) { + ret = bind_at(ws, bo, window, va); + if (!ret) { + if (va + bo->size > ws->next[window]) + ws->next[window] = va + bo->size; + return 0; + } + if (ret != -EEXIST) + return ret; + } + } + return -ENOSPC; +} + +/* Give an unbound buffer back to the kernel. The mapping goes with it: + * nothing else drops it, so a client that allocates and frees - a display + * server does it per pixmap - exhausts its address space rather than its + * memory. */ +static void bo_close(struct sgx_winsys *ws, struct sgx_bo *bo) +{ + struct drm_gem_close c; + + if (bo->map && ws->ops.munmap) + ws->ops.munmap(ws->ops.dev, bo->map, (size_t)bo->size); + + memset(&c, 0, sizeof c); + c.handle = bo->handle; + (void)ws->ops.ioctl(ws->ops.dev, DRM_IOCTL_GEM_CLOSE, &c); + memset(bo, 0, sizeof(*bo)); +} + +void sgx_bo_free(struct sgx_winsys *ws, struct sgx_bo *bo) +{ + unsigned i; + + if (!ws || !bo || !bo->handle) + return; + /* Named by a promised blit, and about to be parked or closed. */ + twod_flush_named(ws, bo, "free"); + + if (bo->gpu_va) { + struct drm_sgx_vm_bind a; + + memset(&a, 0, sizeof a); + a.handle = bo->handle; + a.flags = SGX_BIND_UNBIND; + a.gpu_va = bo->gpu_va; + a.size = bo->size; + a.window = bo->window; + /* Unbind first: the kernel holds a reference for the binding, + * so closing the handle without it leaks the object rather + * than freeing it. */ + (void)ws->ops.ioctl(ws->ops.dev, DRM_IOCTL_SGX_VM_BIND, &a); + range_drop(ws, bo->window, bo->gpu_va); + } + + /* Park it rather than give it back, unless the cache is full or + * already holding enough. + * + * A parked buffer keeps its mapping and the cache must not take it + * away: the pointer can still be held elsewhere - the context's own + * vertex buffer entries name it, and the draw module reads the + * caller's arrays through it - so an entry closed to make room for + * another faulted in JIT-compiled shader code reading a vertex array + * that had just been unmapped. That is why the cache is bounded by + * its slots and its byte cap and never evicts: the winsys cannot know + * what still points into a mapping it handed out. The binding has just been dropped above and + * the mapping is kept, so what comes back out is what a fresh buffer + * looks like - unbound, with an address the caller chooses - which is + * what callers expect: they bind only when gpu_va is zero, and a + * parked binding left them naming an address the frame never meant. + * + * SGX_NO_BO_CACHE gives the buffer straight back, as this did before. */ + if (!SGX_ENVS_WS("SGX_NO_BO_CACHE") && + ws->cached_bytes + bo->size <= SGX_BO_CACHE_BYTES) + for (i = 0; i < SGX_BO_CACHE_NUM; i++) + if (!ws->cache[i].valid) { + ws->cache[i].bo = *bo; + ws->cache[i].bo.gpu_va = 0; + ws->cache[i].bo.window = 0; + /* Nothing outstanding means nothing can still + * be reading it, so it is free at once; + * otherwise it waits for a completion. */ + ws->cache[i].safe_after = ws->busy ? + ws->completed + 1u : + 0u; + ws->cache[i].valid = 1; + ws->cached_bytes += bo->size; + memset(bo, 0, sizeof *bo); + return; + } + + bo_close(ws, bo); +} + +/* Hand back what the cache holds. The kernel closes handles only when the + * file descriptor goes, so a winsys destroyed with entries parked leaked + * their objects and mappings for the rest of the process: destroy calls + * this, and so may a caller under memory pressure. */ +void sgx_bo_cache_drain(struct sgx_winsys *ws) +{ + unsigned i; + + if (!ws) + return; + for (i = 0; i < SGX_BO_CACHE_NUM; i++) { + if (!ws->cache[i].valid) + continue; + ws->cached_bytes -= ws->cache[i].bo.size; + ws->cache[i].valid = 0; + bo_close(ws, &ws->cache[i].bo); + } +} + +unsigned sgx_bo_cache_count(const struct sgx_winsys *ws) +{ + unsigned i, n = 0; + + if (!ws) + return 0; + for (i = 0; i < SGX_BO_CACHE_NUM; i++) + n += ws->cache[i].valid; + return n; +} + +int sgx_submit_oom(struct sgx_winsys *ws, const uint32_t *ta, uint32_t ta_dwords, + const uint32_t *raster, uint32_t raster_dwords, + const uint32_t *oom, uint32_t oom_dwords, + struct sgx_bo *const *bos, uint32_t nbo, + uint32_t width, uint32_t height, int *out_fence) +{ + struct drm_sgx_submit a; + uint32_t *handles; + uint32_t i; + int ret; + + if (!ws || !ta || !ta_dwords || !bos || !nbo) + return -EINVAL; + /* Every object a frame names has to be bound before it is submitted: + * the kernel has no relocations to fix up an unbound one, so an + * address in the stream that nothing backs is a fault rather than a + * patch. Catch it here, where the caller can still be told which. */ + for (i = 0; i < nbo; i++) + if (!bos[i] || !bos[i]->gpu_va) + return -EINVAL; + + handles = calloc(nbo, sizeof *handles); + if (!handles) + return -ENOMEM; + for (i = 0; i < nbo; i++) + handles[i] = bos[i]->handle; + + /* The ordering rule, kept: every blit promised before this frame has + * run before the frame does. The kernel drains a render before it + * runs a job, and this drains the jobs before a render - so the two + * never overlap, whatever memory they name. */ + twod_flush_all(ws, "submit"); + + memset(&a, 0, sizeof a); + a.ta_stream = (uint64_t)(uintptr_t)ta; + a.ta_stream_count = ta_dwords; + a.raster_stream = (uint64_t)(uintptr_t)raster; + a.raster_stream_count = raster_dwords; + a.bo_handles = (uint64_t)(uintptr_t)handles; + a.bo_count = nbo; + a.oom_stream = (uint64_t)(uintptr_t)oom; + a.oom_stream_count = oom_dwords; + a.width = width; + a.height = height; + /* No fence is asked for: nothing here waits on one, and asking makes + * the kernel wait in line for the render so it can hand out a fence + * that is already signalled - which is exactly what this is avoiding. + * sgx_wait_idle() is the wait instead. */ + a.in_sync_fd = -1; + a.out_sync_fd = -1; + if (ws->async) + a.flags |= SGX_SUBMIT_DEFER_WAIT; + /* Cleared whether or not the submit succeeds: the arming belongs to + * the frame that has just been handed over, and a frame that was + * refused took its armed objects with it. */ + if (ws->vistest_next) { + a.flags |= SGX_SUBMIT_VISTEST; + ws->vistest_next = 0; + } + if (ws->msaa_axis == 2) + a.flags |= SGX_SUBMIT_MSAA_2X2; + ret = ws->ops.ioctl(ws->ops.dev, DRM_IOCTL_SGX_SUBMIT, &a); + if (!ret && out_fence) + *out_fence = a.out_sync_fd; + /* Only on success: a submit that failed fired nothing, and the kernel + * has already drained whatever was outstanding before it refused. */ + if (!ret && ws->async) { + ws->busy = 1; + ws->deferred++; + } + free(handles); + return ret; +} + +int sgx_submit(struct sgx_winsys *ws, const uint32_t *ta, uint32_t ta_dwords, + const uint32_t *raster, uint32_t raster_dwords, + struct sgx_bo *const *bos, uint32_t nbo, + uint32_t width, uint32_t height, int *out_fence) +{ + return sgx_submit_oom(ws, ta, ta_dwords, raster, raster_dwords, + NULL, 0, bos, nbo, width, height, out_fence); +} + +int sgx_winsys_lost(const struct sgx_winsys *ws) +{ + return ws ? ws->lost : 0; +} + +/* ---- the 2D block ---- */ + +/* Drain this CPU's write-combining buffers. + * + * Every surface the engine reads was filled through a write-combining + * mapping, and a write-combining buffer is emptied by a store fence on the + * CPU that filled it. The kernel fences before it pushes the stream, but by + * then the thread may be on another CPU and the buffer holding the last of + * the pixels is on the one it left - which is the reasoning sgx_flush() + * already carries for a frame's own contents, and the 2D path had no fence at + * all. A source the CPU wrote and the engine then read as zeros is what that + * costs; XPutImage is the case with no render behind it, so nothing else + * could have made those bytes visible. + * + * Cheap enough to do at both ends: once where the caller's writes just + * happened, and once where the job is handed over. */ +static void twod_store_fence(void) +{ +#if defined(__i386__) || defined(__x86_64__) + __builtin_ia32_sfence(); +#else + __sync_synchronize(); +#endif +} + +struct sgx_twod { + struct sgx_winsys *ws; + uint32_t *w; /* the stream, grown as blits are added */ + unsigned n, cap; + /* Every object the stream names, by GEM handle. Not by struct address: + * the driver above holds one object through several struct sgx_bo of + * its own - the context copies the render target and each texture slot + * by value, and the buffer cache parks and hands back copies - so a + * pointer match misses the very rebind and free twod_flush_named() is + * there to catch, and can match a struct that has since been reused + * for a different handle. The handle is what the kernel is given and + * what identifies the object. */ + struct { uint32_t handle; } *named; + unsigned nh, hcap; + unsigned blits; + /* What the engine currently has bound, so a surface is described once + * per run of blits into it rather than once per blit, and the offset + * each was described at - the address is what the stream carries, and + * an object that moved between two blits of one job needs describing + * again however alike the two surfaces look. */ + struct sgx_twod_surf dst, src; + uint32_t dst_off, src_off; + int dst_set, src_set; + /* Blits per job before it is submitted on its own: SGX_2D_BATCH, and + * unlimited by default. Multi-blit streams are the vendor's shape; + * this is the knob that tests one blit per job on hardware. */ + unsigned batch; +}; + +int sgx_twod_available(struct sgx_winsys *ws) +{ + struct drm_sgx_get_param p; + + if (!ws) + return 0; + if (ws->twod < 0) { + memset(&p, 0, sizeof p); + p.param = SGX_PARAM_TWOD_BASE; + ws->twod = !ws->ops.ioctl(ws->ops.dev, DRM_IOCTL_SGX_GET_PARAM, + &p) && !SGX_ENVS_WS("SGX_NO_2D"); + /* Whatever the kernel says, rather than a copy of it: the base + * is the kernel's own page, outside every window, and a + * surface is reachable when it falls inside the 28-bit offset + * field from there - which twod_surf_offset() is what tests. */ + ws->twod_base = p.value; + /* Said once, and not only under SGX_DEBUG. Whether the block + * is in use is a variable experiments are controlled on, and + * an environment variable set on the wrong process reads as a + * control that is not one - the same trap + * LIBGL_ALWAYS_SOFTWARE=1 on an X server is. This line is in + * the log of the process that actually reads the switch. */ + fprintf(sgx_ws_log(), + "sgx: 2D block %s (SGX_NO_2D %s, base 0x%llx, reach to 0x%llx)\n", + ws->twod > 0 ? "in use" : "not in use", + SGX_ENVS_WS("SGX_NO_2D") ? "set" : "unset", + (unsigned long long)ws->twod_base, + (unsigned long long)(ws->twod_base + SGX_2D_REACH)); + } + return ws->twod > 0; +} + +uint64_t sgx_twod_base(struct sgx_winsys *ws) +{ + return sgx_twod_available(ws) ? ws->twod_base : 0; +} + +struct sgx_twod *sgx_twod_begin(struct sgx_winsys *ws) +{ + struct sgx_twod *t; + const char *e; + + if (!sgx_twod_available(ws)) + return NULL; + t = calloc(1, sizeof *t); + if (!t) + return NULL; + t->ws = ws; + e = getenv("SGX_2D_BATCH"); + t->batch = e && *e ? (unsigned)strtoul(e, NULL, 0) : 0u; + return t; +} + +static int twod_room(struct sgx_twod *t, unsigned dwords) +{ + if (t->n + dwords > t->cap) { + unsigned cap = t->cap ? t->cap * 2u : 64u; + uint32_t *w; + + while (cap < t->n + dwords) + cap *= 2u; + w = realloc(t->w, cap * sizeof *w); + if (!w) + return -ENOMEM; + t->w = w; + t->cap = cap; + } + return 0; +} + +static int twod_name(struct sgx_twod *t, const struct sgx_bo *bo) +{ + unsigned i; + + if (!bo || !bo->handle) + return -EINVAL; + for (i = 0; i < t->nh; i++) + if (t->named[i].handle == bo->handle) + return 0; + if (t->nh == t->hcap) { + unsigned cap = t->hcap ? t->hcap * 2u : 8u; + void *h = realloc(t->named, cap * sizeof *t->named); + + if (!h) + return -ENOMEM; + t->named = h; + t->hcap = cap; + } + t->named[t->nh].handle = bo->handle; + t->nh++; + return 0; +} + +static int twod_names(const struct sgx_twod *t, const struct sgx_bo *bo) +{ + unsigned i; + + if (!bo || !bo->handle) + return 0; + for (i = 0; t && i < t->nh; i++) + if (t->named[i].handle == bo->handle) + return 1; + return 0; +} + +/* The offset a surface word carries, or -ERANGE for one the engine cannot + * reach: unbound, in another window, above the 28-bit field, or with rows + * past the object. rows is how many the caller will touch from the top. */ +static int twod_surf_offset(const struct sgx_winsys *ws, + const struct sgx_twod_surf *s, unsigned rows, + uint32_t *out) +{ + uint64_t at; + + if (!s || !s->bo || !s->bo->gpu_va || s->bo->window != SGX_VM_SURFACE) + return -ERANGE; + at = s->bo->gpu_va + s->offset; + if (at < ws->twod_base || at - ws->twod_base >= SGX_2D_REACH) + return -ERANGE; + if (!s->stride || (s->stride & 3u) || + s->stride > SGX_2D_SURF_STRIDE_MASK || (s->offset & 3u)) + return -EINVAL; + if (!sgx_2d_format_bpp(s->format)) + return -EINVAL; + if ((uint64_t)s->offset + (uint64_t)rows * s->stride > s->bo->size) + return -EINVAL; + *out = (uint32_t)(at - ws->twod_base); + return 0; +} + +static int twod_same_surf(const struct sgx_twod_surf *a, + const struct sgx_twod_surf *b) +{ + return a->bo == b->bo && a->offset == b->offset && + a->stride == b->stride && a->format == b->format; +} + +static int twod_submit(struct sgx_twod *t); + +/* Describe a surface to the engine if it is not the one it has. The vendor + * driver puts a FENCE before every blit after the first and before a + * surface it re-describes (psb_accel.c:752, :801, :902); the same shape here. */ +static int twod_bind(struct sgx_twod *t, unsigned bh, + const struct sgx_twod_surf *s, uint32_t off) +{ + struct sgx_twod_surf *cur = bh == SGX_2D_DST_SURF_BH ? &t->dst : &t->src; + uint32_t *curoff = bh == SGX_2D_DST_SURF_BH ? &t->dst_off : &t->src_off; + int *set = bh == SGX_2D_DST_SURF_BH ? &t->dst_set : &t->src_set; + int ret; + + if (*set && *curoff == off && twod_same_surf(cur, s)) + return 0; + ret = twod_room(t, 2); + if (ret) + return ret; + /* Named before it is described, so a job that cannot record the object + * has not already written the block that reaches its memory. */ + ret = twod_name(t, s->bo); + if (ret) + return ret; + t->w[t->n++] = sgx_2d_surf_word(bh, s->format, s->stride); + t->w[t->n++] = off & SGX_2D_ADDR_MASK; + *cur = *s; + *curoff = off; + *set = 1; + return 0; +} + +/* A job that has grown to the kernel's limit, or to the batch size, goes on + * its own; the caller's next blit starts a fresh one. */ +static int twod_make_room(struct sgx_twod *t, unsigned dwords) +{ + if (t->n + dwords + 1u + SGX_2D_TRAILER_DWORDS > SGX_2D_MAX_DWORDS || + (t->batch && t->blits >= t->batch)) + return twod_submit(t); + return 0; +} + +static int twod_fence(struct sgx_twod *t) +{ + int ret = twod_room(t, 1); + + if (ret) + return ret; + if (t->blits) + t->w[t->n++] = SGX_2D_FENCE_BH; + return 0; +} + +/* Where a job stood before a blit started appending to it, so a blit that + * cannot be completed leaves nothing behind. A surface block with no blit + * after it is not inert: it describes what the engine draws into next, and + * the job carries it into whatever is appended after. */ +struct twod_mark { + unsigned n, nh; + struct sgx_twod_surf dst, src; + uint32_t dst_off, src_off; + int dst_set, src_set; +}; + +static void twod_mark_take(const struct sgx_twod *t, struct twod_mark *m) +{ + m->n = t->n; + m->nh = t->nh; + m->dst = t->dst; + m->src = t->src; + m->dst_off = t->dst_off; + m->src_off = t->src_off; + m->dst_set = t->dst_set; + m->src_set = t->src_set; +} + +static void twod_rewind(struct sgx_twod *t, const struct twod_mark *m) +{ + t->n = m->n; + t->nh = m->nh; + t->dst = m->dst; + t->src = m->src; + t->dst_off = m->dst_off; + t->src_off = m->src_off; + t->dst_set = m->dst_set; + t->src_set = m->src_set; +} + +int sgx_twod_copy(struct sgx_twod *t, const struct sgx_twod_surf *dst, + int dx, int dy, const struct sgx_twod_surf *src, + int sx, int sy, int w, int h) +{ + uint32_t doff, soff, order; + struct twod_mark mark; + int ret; + + if (!t || !dst || !src || w <= 0 || h <= 0 || dx < 0 || dy < 0 || + sx < 0 || sy < 0) + return -EINVAL; + if (dst->format != src->format) + return -EINVAL; + if (dx + w > (int)SGX_2D_COORD_MAX + 1 || + dy + h > (int)SGX_2D_COORD_MAX + 1 || + sx + w > (int)SGX_2D_COORD_MAX + 1 || + sy + h > (int)SGX_2D_COORD_MAX + 1) + return -EINVAL; + ret = twod_surf_offset(t->ws, dst, (unsigned)(dy + h), &doff); + if (ret) + return ret; + ret = twod_surf_offset(t->ws, src, (unsigned)(sy + h), &soff); + if (ret) + return ret; + ret = twod_make_room(t, 2 + 2 + 1 + 4); + if (ret) + return ret; + twod_mark_take(t, &mark); + /* Within one surface the walk has to start at the far corner when the + * destination is below or to the right of the source, or the copy + * reads what it has already written - psbAccelCopyDirection() at + * psb_accel.c:396 and the overlap fix at :990-1006. */ + order = SGX_2D_COPYORDER_TL2BR; + if (twod_same_surf(dst, src) && dx < sx + w && sx < dx + w && + dy < sy + h && sy < dy + h) { + int xback = dx > sx, yback = dy > sy; + + order = xback ? (yback ? SGX_2D_COPYORDER_BR2TL : + SGX_2D_COPYORDER_TR2BL) + : (yback ? SGX_2D_COPYORDER_BL2TR : + SGX_2D_COPYORDER_TL2BR); + if (xback) { + sx += w - 1; + dx += w - 1; + } + if (yback) { + sy += h - 1; + dy += h - 1; + } + } + ret = twod_fence(t); + if (!ret) + ret = twod_bind(t, SGX_2D_DST_SURF_BH, dst, doff); + if (!ret) + ret = twod_bind(t, SGX_2D_SRC_SURF_BH, src, soff); + if (!ret) + ret = twod_room(t, 5); + if (ret) { + twod_rewind(t, &mark); + return ret; + } + t->w[t->n++] = SGX_2D_SRC_OFF_BH | SGX_2D_XY(sx, sy); + t->w[t->n++] = SGX_2D_BLIT_BH | SGX_2D_ROT_NONE | order | + SGX_2D_USE_FILL | SGX_2D_ROP3(SGX_2D_ROP_SRCCOPY); + t->w[t->n++] = 0; /* the fill word USE_FILL asks for */ + t->w[t->n++] = SGX_2D_XY(dx, dy); + t->w[t->n++] = SGX_2D_XY(w, h); + t->blits++; + return 0; +} + +int sgx_twod_fill(struct sgx_twod *t, const struct sgx_twod_surf *dst, + int x, int y, int w, int h, uint32_t argb) +{ + uint32_t doff; + struct twod_mark mark; + int ret; + + if (!t || !dst || w <= 0 || h <= 0 || x < 0 || y < 0) + return -EINVAL; + if (x + w > (int)SGX_2D_COORD_MAX + 1 || y + h > (int)SGX_2D_COORD_MAX + 1) + return -EINVAL; + ret = twod_surf_offset(t->ws, dst, (unsigned)(y + h), &doff); + if (ret) + return ret; + ret = twod_make_room(t, 2 + 4); + if (ret) + return ret; + twod_mark_take(t, &mark); + ret = twod_fence(t); + if (!ret) + ret = twod_bind(t, SGX_2D_DST_SURF_BH, dst, doff); + if (!ret) + ret = twod_room(t, 4); + if (ret) { + twod_rewind(t, &mark); + return ret; + } + t->w[t->n++] = SGX_2D_BLIT_BH | SGX_2D_ROT_NONE | + SGX_2D_COPYORDER_TL2BR | SGX_2D_USE_FILL | + SGX_2D_ROP3(SGX_2D_ROP_PATCOPY); + t->w[t->n++] = argb; + t->w[t->n++] = SGX_2D_XY(x, y); + t->w[t->n++] = SGX_2D_XY(w, h); + t->blits++; + return 0; +} + +unsigned sgx_twod_count(const struct sgx_twod *t) +{ + return t ? t->blits : 0u; +} + +/* Hand the stream to the kernel and start the job over. */ +static int twod_submit(struct sgx_twod *t) +{ + struct drm_sgx_blit a; + struct sgx_winsys *ws = t->ws; + uint32_t *handles; + unsigned i; + int ret; + + /* Nothing to run, and nothing may be left behind either: dwords with + * no blit among them are a surface description the next job would + * open with, naming an address that is a flush old. */ + if (!t->blits) { + t->n = 0; + t->nh = 0; + t->dst_set = t->src_set = 0; + return 0; + } + handles = calloc(t->nh ? t->nh : 1u, sizeof *handles); + if (!handles) + return -ENOMEM; + for (i = 0; i < t->nh; i++) + handles[i] = t->named[i].handle; + memset(&a, 0, sizeof a); + a.stream = (uint64_t)(uintptr_t)t->w; + a.stream_count = t->n; + a.bo_handles = (uint64_t)(uintptr_t)handles; + a.bo_count = t->nh; + a.in_sync_fd = -1; + a.out_sync_fd = -1; + twod_store_fence(); + /* Deferred when asked for: the ioctl then comes back once the stream + * is on the engine, and the kernel drains it wherever it drains a + * deferred render - the next submit, a bind, a wait. Off unless + * SGX_2D_ASYNC asks, because the queue is what removed the round trip + * per copy and this only removes the one per flush. */ + if (SGX_ENVS_WS("SGX_2D_ASYNC")) + a.flags |= SGX_BLIT_DEFER_WAIT; + ret = ws->ops.ioctl(ws->ops.dev, DRM_IOCTL_SGX_BLIT, &a); + if (sgx_ws_debug()) + fprintf(sgx_ws_log(), "sgx: 2D job: %u blit(s), %u dword(s), %u " + "object(s) -> %d\n", t->blits, t->n, t->nh, ret); + /* A job the kernel ran drained the deferred render on the way in, so + * what was outstanding is done and the next map or rebind need not + * wait for it again. + * + * Only a job it ran. The argument checks at the head of the ioctl + * come before that drain, so a refused job has drained nothing - and + * recording it as drained is worse than doing nothing at all: it + * takes ws->busy down, and with it every later sgx_wait_idle() and + * every sgx_bo_is_busy(). The guard in sgx_use_target() that keeps a + * render target from moving out from under a running render is one of + * those, and what it prevents is a copy that comes back black. Left + * set, the cost is one wait that finds nothing to do. */ + if (!ret && ws->busy) { + if (getenv("SGX_ASYNC_STATS")) + wait_count(ws, "sgx_twod"); + ws->busy = 0; + ws->completed++; + } + if (!ret) { + ws->twod_jobs++; + ws->twod_blits += t->blits; + if (a.flags & SGX_BLIT_DEFER_WAIT) + ws->twod_busy = 1; + } + free(handles); + t->n = 0; + t->nh = 0; + t->blits = 0; + t->dst_set = t->src_set = 0; + return ret; +} + +int sgx_twod_end(struct sgx_twod *t) +{ + int ret; + + if (!t) + return -EINVAL; + ret = twod_submit(t); + sgx_twod_abort(t); + return ret; +} + +void sgx_twod_abort(struct sgx_twod *t) +{ + if (!t) + return; + free(t->w); + free(t->named); + free(t); +} + +unsigned sgx_winsys_lost_gen(const struct sgx_winsys *ws) +{ + return ws ? ws->lost_gen : 0; +} + +/* ---- the open job ---- + * + * The blits a caller promises go here and stay until something has to see + * them. What forces a flush is not a policy choice: the stream carries the + * addresses the objects had when the blit was appended, and the result is not + * in memory until the engine has run - so every hook is one of "this object + * is about to move or go" (twod_flush_named, from bind, unbind and free), + * "someone is about to look at memory" (sgx_wait_idle) or "a render is about + * to run" (sgx_submit_oom). The Gallium driver adds the two Mesa asks for by + * name, flush and flush_resource. + */ +static struct sgx_twod *twod_open(struct sgx_winsys *ws) +{ + if (!ws->job) + ws->job = sgx_twod_begin(ws); + return ws->job; +} + +static int twod_flush_all(struct sgx_winsys *ws, const char *why) +{ + int ret; + + if (!ws || !ws->job || !sgx_twod_count(ws->job)) + return 0; + ws->twod_flushes++; + if (sgx_ws_debug()) + fprintf(sgx_ws_log(), "sgx: 2D flush (%s): %u blit(s)\n", + why ? why : "?", sgx_twod_count(ws->job)); + ret = twod_submit(ws->job); + if (ret) + fprintf(sgx_ws_log(), "sgx: a 2D job of %u blit(s) was lost: " + "%d\n", ws->twod_queued, ret); + ws->twod_queued = 0; + return ret; +} + +static void twod_flush_named(struct sgx_winsys *ws, const struct sgx_bo *bo, + const char *why) +{ + if (ws && ws->job && bo && twod_names(ws->job, bo)) + (void)twod_flush_all(ws, why); +} + +int sgx_twod_flush(struct sgx_winsys *ws) +{ + return twod_flush_all(ws, "caller"); +} + +unsigned sgx_twod_pending(const struct sgx_winsys *ws) +{ + return ws && ws->job ? sgx_twod_count(ws->job) : 0u; +} + +int sgx_twod_queue_copy(struct sgx_winsys *ws, const struct sgx_twod_surf *dst, + int dx, int dy, const struct sgx_twod_surf *src, + int sx, int sy, int w, int h) +{ + struct sgx_twod *t = twod_open(ws); + int ret; + + if (!t) + return -ENODEV; + ret = sgx_twod_copy(t, dst, dx, dy, src, sx, sy, w, h); + if (!ret) { + /* Here, rather than only at the flush: this is the CPU the + * caller's writes were made on, and the flush may be on + * another one an unbounded time later. */ + twod_store_fence(); + ws->twod_queued++; + } + return ret; +} + +int sgx_twod_queue_fill(struct sgx_winsys *ws, const struct sgx_twod_surf *dst, + int x, int y, int w, int h, uint32_t argb) +{ + struct sgx_twod *t = twod_open(ws); + int ret; + + if (!t) + return -ENODEV; + ret = sgx_twod_fill(t, dst, x, y, w, h, argb); + if (!ret) { + twod_store_fence(); + ws->twod_queued++; + } + return ret; +} diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_winsys.h mesa-26.2.2/src/gallium/drivers/sgx/sgx_winsys.h --- mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_winsys.h 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/sgx_winsys.h 2026-09-08 10:57:36.681278060 +0200 @@ -0,0 +1,289 @@ +/* SPDX-License-Identifier: MIT */ +/* + * The userspace side of driver/uapi/sgx_drm.h: buffer objects, addresses and + * submission, with the ioctls behind a vtable so the whole layer can be tested + * on the host against a mock instead of only on hardware. + * + * This is the layer a Gallium driver sits on. It exists before the Gallium + * driver because the UAPI should be exercised by something real before it is + * frozen, and because the frame builder that would feed it + * (tools/xpsb-open/xpsb_frame.c) already works. + * + * Copyright (C) 2026 René Rebe + */ +#ifndef _SGX_WINSYS_H_ +#define _SGX_WINSYS_H_ + +#include +#include + +struct sgx_winsys; + +struct sgx_bo { + uint32_t handle; + uint64_t size; + uint64_t gpu_va; /* 0 until bound */ + uint32_t window; + void *map; + /* The completion count this buffer is idle after. A render already + * fired may still be reading or writing it, and moving its mapping + * out from under the core loses what has not been written yet - so + * whoever wants to move it waits, but only for a buffer the core can + * still be using rather than for the core as a whole. */ + unsigned busy_gen; +}; + +/* Every ioctl the winsys makes, so a test can supply its own. */ +struct sgx_ioctl_ops { + int (*ioctl)(void *dev, unsigned long req, void *arg); + void *(*mmap)(void *dev, uint64_t offset, size_t len); + /* Optional: a winsys without it leaks a mapping per freed object. */ + void (*munmap)(void *dev, void *addr, size_t len); + void *dev; +}; + +struct sgx_winsys *sgx_winsys_create(const struct sgx_ioctl_ops *ops); +void sgx_winsys_destroy(struct sgx_winsys *ws); +/* Whether a render has been lost since this winsys was created. */ +int sgx_winsys_lost(const struct sgx_winsys *ws); +/* How many renders have been lost. A caller that reports a reset once per loss + * compares this with what it last saw; the sticky flag above cannot tell a new + * loss from an old one. */ +unsigned sgx_winsys_lost_gen(const struct sgx_winsys *ws); + +/* Waits that have completed, and a note that this buffer is in the frame + * about to be submitted. Together they say whether moving it has to wait. */ +unsigned sgx_winsys_completed(const struct sgx_winsys *ws); +void sgx_bo_mark_busy(struct sgx_winsys *ws, struct sgx_bo *bo); +int sgx_bo_is_busy(const struct sgx_winsys *ws, const struct sgx_bo *bo); + +/* The device the winsys was made from, for the callers that have to hand it + * back - a pipe_screen is asked for its file descriptor. Negative if the + * winsys is a test's rather than a device's. */ +int sgx_winsys_fd(struct sgx_winsys *ws); +void sgx_winsys_set_fd(struct sgx_winsys *ws, int fd); + +int sgx_bo_new(struct sgx_winsys *ws, struct sgx_bo *bo, uint64_t size, + uint32_t flags); +int sgx_bo_map(struct sgx_winsys *ws, struct sgx_bo *bo); + +/* + * Address assignment is the winsys's job, not the kernel's - that is what + * makes the submit path relocation free. Each window is a bump allocator, + * which is enough for one frame at a time and is where a real allocator goes. + */ +int sgx_bo_bind(struct sgx_winsys *ws, struct sgx_bo *bo, uint32_t window); + +/* Bind at an address the caller chooses rather than the next free one. + * + * The generated frame streams carry absolute GPU addresses - the heap at + * 0x20010000, the USSE code at 0x20800000, and so on - so anything replaying + * such a stream has to put its objects exactly there. Bump-allocating would + * put the second object immediately after the first and the stream would + * reference nothing. */ +/* Ask the kernel which USE base register covers a range of code, and the + * offset within it. Relocations of USE_REG/USE_OFFSET records need both. */ +/* The console framebuffer as a render target: the kernel maps it at the + * render-target window and reports its geometry. Nothing is allocated and + * nothing is copied. */ +struct sgx_scanout { + uint64_t gpu_va; + uint32_t pitch, width, height, size; +}; + +int sgx_get_scanout(struct sgx_winsys *ws, struct sgx_scanout *out); + +int sgx_use_base(struct sgx_winsys *ws, uint64_t gpu_va, uint32_t size, + uint32_t data_master, uint32_t *reg, uint32_t *offset); + +/* Take a binding back, so the address it held can be given to another object. + * A render target's address is fixed by the frame's relocations, so which + * object is at it changes as the bound framebuffer does. */ +int sgx_bo_unbind(struct sgx_winsys *ws, struct sgx_bo *bo); + +/* Keep the bump allocator off a range. Some objects have addresses the frame's + * relocations already name - the render target, the depth buffer and the + * frame's own texture all sit at fixed places in the surface window - so an + * allocation that started at the window's beginning would be handed one of + * them. Call it once, before anything is allocated in that window. */ +int sgx_winsys_reserve(struct sgx_winsys *ws, uint32_t window, uint64_t upto); + +/* How large a window is, so a caller sizing an allocation against it does not + * have to carry its own copy of the table. Zero for a window that does not + * exist. */ +uint64_t sgx_window_bytes(uint32_t window); + +int sgx_bo_bind_at(struct sgx_winsys *ws, struct sgx_bo *bo, uint32_t window, + uint64_t gpu_va); + +/* The out-of-memory stream goes with every frame, not only one that runs out: + * the fire path branches on having been given one. */ +int sgx_submit_oom(struct sgx_winsys *ws, const uint32_t *ta, uint32_t ta_dwords, + const uint32_t *raster, uint32_t raster_dwords, + const uint32_t *oom, uint32_t oom_dwords, + struct sgx_bo *const *bos, uint32_t nbo, + uint32_t width, uint32_t height, int *out_fence); + +int sgx_submit(struct sgx_winsys *ws, const uint32_t *ta, uint32_t ta_dwords, + const uint32_t *raster, uint32_t raster_dwords, + struct sgx_bo *const *bos, uint32_t nbo, + uint32_t width, uint32_t height, int *out_fence); + +uint64_t sgx_window_start(uint32_t window); +uint64_t sgx_window_size(uint32_t window); + +/* Release a buffer object: unbind it if it is bound, then either park it in + * the winsys' cache for the next allocation of its size or, when the cache is + * full or SGX_NO_BO_CACHE is set, close the handle. A parked object is still + * the winsys' to close: the drain does that for every entry, and destroying + * the winsys drains it, so nothing outlives the winsys but what a caller + * still holds. The count is for callers that want to account for what is + * parked - a test that checks nothing leaked asks it, or drains and counts + * the closes. + */ +void sgx_bo_free(struct sgx_winsys *ws, struct sgx_bo *bo); +void sgx_bo_cache_drain(struct sgx_winsys *ws); +unsigned sgx_bo_cache_count(const struct sgx_winsys *ws); + +/* + * Deferred waiting. + * + * With it on, the submit ioctl comes back once the render is fired rather than + * once it has ended, so the application builds the next frame while the core + * is still on this one. The wait does not go away, it moves: the kernel drains + * it at the head of the next submit, and sgx_wait_idle() is where userspace + * drains it before it touches anything the render may still be reading. + * + * That makes every caller of this layer responsible for one rule: nothing may + * read or overwrite GPU-visible memory between a submit and the next + * sgx_wait_idle(). sgx_resource_map() and sgx_flush() are where that is + * honoured, so a caller going through them does not have to think about it. + * + * Off unless asked for, and never on a winsys without a real device behind it. + */ +void sgx_winsys_set_async(struct sgx_winsys *ws, int on); + +/* Multisampling for the frames that follow. samples is XPSB_MSAA_1X or + * XPSB_MSAA_4X; anything else is refused with -EINVAL rather than rounded, + * because the kernel sizes the scene's region-header array from this and a + * count it did not expect has the tiler write past the end of it. */ +int sgx_winsys_set_msaa(struct sgx_winsys *ws, unsigned samples); +unsigned sgx_winsys_msaa(const struct sgx_winsys *ws); + +/* Wait for a deferred render to end. A no-op when none is outstanding. + * + * The site name is not decoration: how much a frame actually overlaps is + * exactly the question of which of these ten places the wait lands in, and + * that cannot be reasoned about from the source - SGX_ASYNC_STATS=1 prints the + * tally when the winsys goes. The macro fills it in, so callers write + * sgx_wait_idle(ws) as they would any other. */ +int sgx_wait_idle_where(struct sgx_winsys *ws, const char *where); +#define sgx_wait_idle(ws) sgx_wait_idle_where((ws), __func__) + +/* + * The ISP's visibility counters, which is what an occlusion query counts. + * + * Eight of them and they are device registers on this core, so the kernel is + * what reads them: it harvests all eight when a render armed for them ends, + * adds them into a per-file running total and clears the registers - the same + * accumulate-and-clear the vendor's microkernel does per render + * (3d.asm:714-776). What comes back here is that total, so a caller reads it + * once before a query and once after and takes the difference; a query that + * spanned several frames has every one of them in it. + * + * A frame is armed by sgx_vistest_arm(), which the next submit consumes: a + * frame that has no visibility-tested object costs the kernel nothing. + */ +#define SGX_VISTEST_COUNTERS 8 + +void sgx_vistest_arm(struct sgx_winsys *ws); +/* The totals. With wait, drains a deferred render first, so what comes back + * includes the frame the last submit fired; without it, *pending says a render + * of this file's is still outstanding and a count in it has not landed. + * counts[] takes SGX_VISTEST_COUNTERS entries. */ +int sgx_vistest_read(struct sgx_winsys *ws, int wait, uint64_t *counts, + int *pending); + +/* A winsys on a real DRM file descriptor: the ioctl and mmap the tests + * inject, done for real. Kept separate from sgx_winsys_create() so the tests + * can supply their own without this file ever opening anything. */ +struct sgx_winsys *sgx_winsys_create_drm(int fd); + +/* + * The 2D block: copies and solid fills without a scene, a shader or the + * tiler. A job is built here in the engine's own block-header format + * (kernel/sgx_twod.h) and handed to DRM_IOCTL_SGX_BLIT, which checks every + * block, appends its own completion and returns once the engine has finished + * - so a job is complete when sgx_twod_end() comes back, and the kernel has + * drained any deferred render before running it. + * + * A surface is a bound object in the surface window, an offset into it, a + * byte stride and a 2D format code. The engine reaches the first 256 MiB of + * that window: the offset field is 28 bits from BIF_TWOD_REQ_BASE. + */ +struct sgx_twod_surf { + const struct sgx_bo *bo; + uint32_t offset; /* bytes into the object */ + uint32_t stride; /* bytes per row, a multiple of 4 */ + uint32_t format; /* SGX_2D_FMT_* */ +}; + +struct sgx_twod; + +/* Nonzero if the kernel has the 2D path. Asked once. */ +int sgx_twod_available(struct sgx_winsys *ws); +/* Where the engine's offsets are based, once available. */ +uint64_t sgx_twod_base(struct sgx_winsys *ws); + +/* A job to add copies and fills to. NULL if the kernel has no 2D path. */ +struct sgx_twod *sgx_twod_begin(struct sgx_winsys *ws); + +/* A w by h copy from (sx, sy) of src to (dx, dy) of dst, both in pixels of + * the surfaces' formats, which must match. Overlap within one surface is + * handled by walking the rectangle in the safe order. -ERANGE if a surface + * is not one the engine can reach, -EINVAL if the rectangle does not fit. */ +int sgx_twod_copy(struct sgx_twod *t, const struct sgx_twod_surf *dst, + int dx, int dy, const struct sgx_twod_surf *src, + int sx, int sy, int w, int h); + +/* Fill w by h at (x, y) with a colour in ARGB8888, which the engine converts + * to the destination's format. */ +int sgx_twod_fill(struct sgx_twod *t, const struct sgx_twod_surf *dst, + int x, int y, int w, int h, uint32_t argb); + +/* Submit what was added and free the job. Zero once the engine has done all + * of it; a job with nothing in it costs no ioctl. */ +int sgx_twod_end(struct sgx_twod *t); +void sgx_twod_abort(struct sgx_twod *t); + +/* Blits added so far, and the dwords they take. */ +unsigned sgx_twod_count(const struct sgx_twod *t); + +/* + * The open job. + * + * A job costs one synchronous ioctl, so a caller that makes one per copy pays + * a round trip per glyph. These append to a job the winsys keeps open and + * hand it to the kernel only when something has to observe the result: a CPU + * map or any other sgx_wait_idle(), a bind, unbind or free of a buffer the + * job names, a 3D frame, or sgx_twod_flush() by name. The ordering rule is + * unchanged - a blit still never overlaps a render that touches the same + * memory - because the frame path flushes the job before it fires and the + * kernel drains a render before it runs a job. + * + * What this costs: a queued blit has been promised, not done. If the job is + * later refused or lost the copy is gone, where an immediate job could still + * have fallen back to the CPU - so everything a job can be refused for is + * checked here, at append time, and a lost job is reported. + */ +int sgx_twod_queue_copy(struct sgx_winsys *ws, const struct sgx_twod_surf *dst, + int dx, int dy, const struct sgx_twod_surf *src, + int sx, int sy, int w, int h); +int sgx_twod_queue_fill(struct sgx_winsys *ws, const struct sgx_twod_surf *dst, + int x, int y, int w, int h, uint32_t argb); +/* Hand the open job to the kernel now. Zero when there was nothing to do. */ +int sgx_twod_flush(struct sgx_winsys *ws); +/* Blits promised and not yet submitted. */ +unsigned sgx_twod_pending(const struct sgx_winsys *ws); + +#endif diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_winsys_drm.c mesa-26.2.2/src/gallium/drivers/sgx/sgx_winsys_drm.c --- mesa-26.2.2.orig/src/gallium/drivers/sgx/sgx_winsys_drm.c 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/sgx_winsys_drm.c 2026-09-08 10:57:36.681289250 +0200 @@ -0,0 +1,62 @@ +/* The real DRM backend for the winsys. + * + * Everything else in this driver reaches the kernel through sgx_ioctl_ops, so + * that the tests can supply a mock and never open a device. This is the one + * place that does the real thing, and it is deliberately the only one. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#include "sgx_winsys.h" + +#include +#include +#include +#include +#include + +static int drm_ioctl(void *dev, unsigned long req, void *arg) +{ + int fd = (int)(intptr_t)dev; + int ret; + + /* DRM ioctls are restarted rather than failed on a signal; returning + * -EINTR to the caller would look like a rejected submit. */ + do { + ret = ioctl(fd, req, arg); + } while (ret == -1 && (errno == EINTR || errno == EAGAIN)); + return ret == -1 ? -errno : 0; +} + +static void *drm_mmap(void *dev, uint64_t offset, size_t len) +{ + int fd = (int)(intptr_t)dev; + void *p = mmap(NULL, len, PROT_READ | PROT_WRITE, MAP_SHARED, fd, + (off_t)offset); + + return p == MAP_FAILED ? NULL : p; +} + +static void drm_munmap(void *dev, void *addr, size_t len) +{ + (void)dev; + munmap(addr, len); +} + +struct sgx_winsys *sgx_winsys_create_drm(int fd) +{ + struct sgx_ioctl_ops ops; + struct sgx_winsys *ws; + + if (fd < 0) + return NULL; + ops.ioctl = drm_ioctl; + ops.mmap = drm_mmap; + ops.munmap = drm_munmap; + ops.dev = (void *)(intptr_t)fd; + ws = sgx_winsys_create(&ops); + if (ws) + sgx_winsys_set_fd(ws, fd); + return ws; +} diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/test_3d.c mesa-26.2.2/src/gallium/drivers/sgx/test_3d.c --- mesa-26.2.2.orig/src/gallium/drivers/sgx/test_3d.c 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/test_3d.c 2026-09-08 10:57:36.684505268 +0200 @@ -0,0 +1,516 @@ +/* Self-test for the SGX/CPU path decision. + * + * Nothing here opens a device. The decision is a pure function of the + * operation, which is the whole point of factoring it out: every rule that + * sends an operation to the CPU fallback - a stride the hardware cannot + * express, a format with no surface code, an operator outside the shader's + * range, a capability the frame layer does not have - is checked on the host, + * where being wrong costs nothing. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#include +#include +#include + +#include "xpsb_3d.h" +#include "xpsb_frame.h" +#include "xpsb_shader.h" +#include "xpsb_yuv.h" + +#define PICT_a8r8g8b8 0x20028888 +#define PICT_x8r8g8b8 0x20020888 +#define PICT_a8b8g8r8 0x20038888 +#define PICT_r5g6b5 0x10020565 +#define PICT_x1r5g5b5 0x10021555 +#define PICT_a8 0x08018000 +#define PICT_a4r4g4b4 0x10024444 + +/* Every capability, so the rules that are not about capabilities can be + * checked on their own. What the module is built with is xpsb_3d_caps(), + * which test_shipped_caps() covers separately. */ +#define ALL_CAPS (XPSB_3D_CAP_QUAD | XPSB_3D_CAP_COMPOSITE | \ + XPSB_3D_CAP_MASK_TEX | XPSB_3D_CAP_SCALAR | \ + XPSB_3D_CAP_VIDEO) + +static int failures; + +static void expect(const char *what, uint32_t got, uint32_t want) +{ + if (got != want) { + printf(" FAIL %-34s got %#010x want %#010x\n", what, got, + want); + failures++; + } +} + +static struct xpsb_3d_surf surf(uint32_t pf, uint32_t w, uint32_t h) +{ + struct xpsb_3d_surf s; + uint32_t fmt = 0; + + memset(&s, 0, sizeof s); + s.handle = 1; + s.pict_format = pf; + s.w = w; + s.h = h; + if (!xpsb_3d_format(pf, &fmt)) + s.stride = xpsb_format_bpp(fmt) * ((w + 31u) & ~31u); + return s; +} + +/* A surface whose hardware format does not come from a PICT format. */ +static struct xpsb_3d_surf raw(uint32_t fmt, uint32_t w, uint32_t h) +{ + struct xpsb_3d_surf s; + + memset(&s, 0, sizeof s); + s.handle = 1; + s.w = w; + s.h = h; + s.stride = xpsb_format_bpp(fmt) * ((w + 31u) & ~31u); + return s; +} + +static struct xpsb_3d_comp good_comp(void) +{ + struct xpsb_3d_comp r; + + memset(&r, 0, sizeof r); + r.op = XPSB_OP_OVER; + r.dst = surf(PICT_a8r8g8b8, 256, 256); + r.src = surf(PICT_a8r8g8b8, 64, 64); + r.have_src = 1; + r.scalar_mask = 1; + return r; +} + +static const float video_conv[XPSB_3D_CONV_N] = { + 1.164f, 0.0f, 1.596f, 1.164f, -0.392f, -0.813f, 1.164f, 2.017f, 0.0f, + -0.0625f, 1.0f +}; + +static struct xpsb_3d_video good_video(void) +{ + struct xpsb_3d_video r; + + memset(&r, 0, sizeof r); + r.fourcc = XPSB_FOURCC_YUY2; + r.nplanes = 1; + r.dst = surf(PICT_a8r8g8b8, 320, 240); + r.plane[0] = raw(XPSB_FMT_YUY2, 320, 240); + r.conv = video_conv; + r.u1 = r.v1 = 1.0f; + return r; +} + +static void test_format(void) +{ + uint32_t f; + + puts("PICT format to hardware surface code:"); + expect("a8r8g8b8", !xpsb_3d_format(PICT_a8r8g8b8, &f) ? f : ~0u, + XPSB_FMT_8888); + expect("x8r8g8b8", !xpsb_3d_format(PICT_x8r8g8b8, &f) ? f : ~0u, + XPSB_FMT_8888); + expect("a8b8g8r8", !xpsb_3d_format(PICT_a8b8g8r8, &f) ? f : ~0u, + XPSB_FMT_BGR8888); + expect("r5g6b5", !xpsb_3d_format(PICT_r5g6b5, &f) ? f : ~0u, + XPSB_FMT_565); + expect("x1r5g5b5", !xpsb_3d_format(PICT_x1r5g5b5, &f) ? f : ~0u, + XPSB_FMT_1555); + expect("a8", !xpsb_3d_format(PICT_a8, &f) ? f : ~0u, XPSB_FMT_A8); + expect("a4r4g4b4 declined", + (uint32_t)xpsb_3d_format(PICT_a4r4g4b4, &f), (uint32_t)-1); + expect("24 bpp declined", (uint32_t)xpsb_3d_format(0x18020888, &f), + (uint32_t)-1); + expect("type 0 declined", (uint32_t)xpsb_3d_format(0x20008888, &f), + (uint32_t)-1); +} + +static void test_surfaces(void) +{ + struct xpsb_3d_surf s; + + puts("surface acceptance:"); + s = surf(PICT_a8r8g8b8, 100, 100); + expect("stride bpp*ALIGN(w,32)", s.stride, 4 * 128); + expect("dest accepted", (uint32_t)xpsb_3d_dest_ok(&s), 0); + + s.stride = 4 * 100; + expect("dest unaligned stride", (uint32_t)xpsb_3d_dest_ok(&s), + (uint32_t)-1); + expect("texture unaligned stride", + (uint32_t)xpsb_3d_tex_ok(&s, XPSB_FMT_8888), (uint32_t)-1); + + s = surf(PICT_a8r8g8b8, 100, 100); + s.handle = 0; + expect("dest without a buffer", (uint32_t)xpsb_3d_dest_ok(&s), + (uint32_t)-1); + + /* Below a tile is fine for both now: the render box counts tiles and a + * target smaller than one still fills one. glamor composites into + * pixmaps of every size, and refusing them dropped whole frames. */ + s = surf(PICT_a8r8g8b8, 8, 8); + expect("dest below one tile fine", + (uint32_t)xpsb_3d_dest_ok(&s), 0); + expect("texture below one tile fine", + (uint32_t)xpsb_3d_tex_ok(&s, XPSB_FMT_8888), 0); + s = surf(PICT_a8r8g8b8, 1, 1); + expect("a single pixel is a target", + (uint32_t)xpsb_3d_dest_ok(&s), 0); + s = surf(PICT_a8r8g8b8, 0, 8); + expect("no width is not", (uint32_t)xpsb_3d_dest_ok(&s), (uint32_t)-1); + + s = surf(PICT_a8r8g8b8, 8192, 64); + expect("dest past 4096", (uint32_t)xpsb_3d_dest_ok(&s), (uint32_t)-1); + + s = surf(PICT_a4r4g4b4, 64, 64); + s.stride = 2 * 64; + expect("dest without a surface code", (uint32_t)xpsb_3d_dest_ok(&s), + (uint32_t)-1); + + s = raw(XPSB_FMT_YUY2, 64, 64); + expect("packed YUV is not a dest", (uint32_t)xpsb_3d_dest_ok(&s), + (uint32_t)-1); + expect("packed YUV is a texture", + (uint32_t)xpsb_3d_tex_ok(&s, XPSB_FMT_YUY2), 0); + + /* Past the hardware's own range, not Xpsb.so's three-mode enum: the + * encoder takes every mode the unit has - mirror, the border forms - + * and every filter mode including the anisotropic pair. */ + s = surf(PICT_a8r8g8b8, 64, 64); + s.umode = 8; + expect("addressing mode out of range", + (uint32_t)xpsb_3d_tex_ok(&s, XPSB_FMT_8888), (uint32_t)-1); + s.umode = XPSB_WRAP_CLAMPGL; + s.magfilter = 4; + expect("filter out of range", + (uint32_t)xpsb_3d_tex_ok(&s, XPSB_FMT_8888), (uint32_t)-1); +} + +static void test_composite(void) +{ + struct xpsb_3d_comp r; + + puts("composite decision, all capabilities present:"); + r = good_comp(); + expect("source texture, scalar mask", + (uint32_t)xpsb_3d_decide_composite(&r, ALL_CAPS), 1); + + /* The range is the shader generator's, 0 to 13. psbExaCheckComposite + * refuses anything above PictOpAdd before it reaches us, so 13 is a + * boundary the DDX never sends, not a case to rely on. */ + r = good_comp(); + r.op = XPSB_OP_CLEAR; + expect("operator 0", (uint32_t)xpsb_3d_decide_composite(&r, ALL_CAPS), + 1); + r.op = XPSB_OP_ADD; + expect("operator 12, the DDX's last", + (uint32_t)xpsb_3d_decide_composite(&r, ALL_CAPS), 1); + r.op = XPSB_NUM_OPS - 1; + expect("operator 13, the shader's last", + (uint32_t)xpsb_3d_decide_composite(&r, ALL_CAPS), 1); + r.op = XPSB_NUM_OPS; + expect("operator 14", (uint32_t)xpsb_3d_decide_composite(&r, ALL_CAPS), + 0); + r.op = -1; + expect("operator -1", (uint32_t)xpsb_3d_decide_composite(&r, ALL_CAPS), + 0); + + r = good_comp(); + r.have_src = 0; + expect("source promised but absent", + (uint32_t)xpsb_3d_decide_composite(&r, ALL_CAPS), 0); + + r = good_comp(); + r.src.stride -= 4; + expect("source stride the hardware cannot express", + (uint32_t)xpsb_3d_decide_composite(&r, ALL_CAPS), 0); + + r = good_comp(); + r.dst.stride -= 4; + expect("dest stride the hardware cannot express", + (uint32_t)xpsb_3d_decide_composite(&r, ALL_CAPS), 0); + + r = good_comp(); + r.src = surf(PICT_a4r4g4b4, 64, 64); + r.src.stride = 2 * 64; + expect("source format with no surface code", + (uint32_t)xpsb_3d_decide_composite(&r, ALL_CAPS), 0); + + r = good_comp(); + r.dst = surf(PICT_a8, 256, 256); + expect("a8 destination", + (uint32_t)xpsb_3d_decide_composite(&r, ALL_CAPS), 1); + r.dst = surf(PICT_r5g6b5, 256, 256); + expect("r5g6b5 destination", + (uint32_t)xpsb_3d_decide_composite(&r, ALL_CAPS), 1); + + r = good_comp(); + r.scalar_src = 1; + r.scalar_mask = 0; + r.have_src = 0; + r.mask = surf(PICT_a8, 64, 64); + r.have_mask = 1; + expect("scalar source with a mask texture", + (uint32_t)xpsb_3d_decide_composite(&r, ALL_CAPS), 1); + /* the mask is the only texture, so it sits on unit 0 and the second + * unit is not wanted - this is how glyph rendering arrives */ + expect("scalar source needs no second unit", + (uint32_t)xpsb_3d_decide_composite(&r, + ALL_CAPS & ~XPSB_3D_CAP_MASK_TEX), + 1); + expect("scalar source without a constant operand", + (uint32_t)xpsb_3d_decide_composite(&r, + ALL_CAPS & ~XPSB_3D_CAP_SCALAR), + 0); + r.mask.stride -= 1; + expect("mask stride the hardware cannot express", + (uint32_t)xpsb_3d_decide_composite(&r, ALL_CAPS), 0); + + r = good_comp(); + r.scalar_mask = 0; + r.mask = surf(PICT_a8, 64, 64); + r.have_mask = 1; + expect("source and mask both textures", + (uint32_t)xpsb_3d_decide_composite(&r, ALL_CAPS), 1); + expect("two textures without a second unit", + (uint32_t)xpsb_3d_decide_composite(&r, + ALL_CAPS & ~XPSB_3D_CAP_MASK_TEX), + 0); + + r = good_comp(); + r.scalar_src = 1; + r.have_src = 0; + expect("both operands scalar", + (uint32_t)xpsb_3d_decide_composite(&r, ALL_CAPS), 0); + + r = good_comp(); + expect("scalar mask without a constant operand", + (uint32_t)xpsb_3d_decide_composite(&r, + ALL_CAPS & ~XPSB_3D_CAP_SCALAR), + 0); + expect("without the composite capability", + (uint32_t)xpsb_3d_decide_composite(&r, + ALL_CAPS & ~XPSB_3D_CAP_COMPOSITE), + 0); + expect("without the quad capability", + (uint32_t)xpsb_3d_decide_composite(&r, + ALL_CAPS & ~XPSB_3D_CAP_QUAD), + 0); + expect("with no capabilities", + (uint32_t)xpsb_3d_decide_composite(&r, 0), 0); +} + +/* Why one sampled unit is enough for either scalar form, and therefore why + * XPSB_3D_CAP_SCALAR does not imply XPSB_3D_CAP_MASK_TEX. The modulate word + * names the operand registers: tools/isa-usse reads 0xb0000000 as + * "sop2 i0, pa0, sa0" and 0xe0000000 as "sop2 i0, sa0, pa0", so the texture + * that is left is pa0 both times - unit 0 - and only two textures reach pa1. */ +static void test_scalar_operands(void) +{ + struct xpsb_shader_flags fl; + uint32_t w[4]; + + puts("scalar operand registers:"); + memset(&fl, 0, sizeof fl); + xpsb_composite_shader(XPSB_OP_OVER, &fl, w); + expect("two textures read pa0 and pa1", w[0], 0xa0000001u); + + fl.scalar_mask = 1; + xpsb_composite_shader(XPSB_OP_OVER, &fl, w); + expect("scalar mask reads pa0 and sa0", w[0], 0xb0000000u); + + memset(&fl, 0, sizeof fl); + fl.scalar_src = 1; + xpsb_composite_shader(XPSB_OP_OVER, &fl, w); + expect("scalar source reads sa0 and pa0", w[0], 0xe0000000u); + + memset(&fl, 0, sizeof fl); + fl.src_is_a8 = 1; + fl.scalar_mask = 1; + xpsb_composite_shader(XPSB_OP_OVER, &fl, w); + expect("a8 source with a scalar mask", w[0], 0xb0000000u); +} + +static void test_video(void) +{ + struct xpsb_3d_video r; + struct xpsb_3d_surf d; + + puts("video decision, all capabilities present:"); + r = good_video(); + expect("YUY2 packed", (uint32_t)xpsb_3d_decide_video(&r, ALL_CAPS), 1); + + r.fourcc = XPSB_FOURCC_UYVY; + r.plane[0] = raw(XPSB_FMT_UYVY, 320, 240); + expect("UYVY packed", (uint32_t)xpsb_3d_decide_video(&r, ALL_CAPS), 1); + + r = good_video(); + r.fourcc = XPSB_FOURCC('X', 'X', 'X', 'X'); + expect("FOURCC with no shader", + (uint32_t)xpsb_3d_decide_video(&r, ALL_CAPS), 0); + + r = good_video(); + r.nplanes = 3; + expect("plane count disagrees with the FOURCC", + (uint32_t)xpsb_3d_decide_video(&r, ALL_CAPS), 0); + + r = good_video(); + r.plane[0].stride -= 2; + expect("plane stride the hardware cannot express", + (uint32_t)xpsb_3d_decide_video(&r, ALL_CAPS), 0); + + /* The planar formats decline on their own plane count, not on + * MASK_TEX: even with every capability and every surface rule + * satisfied, the FIRH programs are undecoded and their coefficient + * registers unnamed. */ + r = good_video(); + r.fourcc = XPSB_FOURCC_NV12; + r.nplanes = 2; + r.plane[0] = raw(XPSB_FMT_A8, 320, 240); + r.plane[1] = raw(XPSB_FMT_A8, 160, 120); + expect("NV12 with every capability", + (uint32_t)xpsb_3d_decide_video(&r, ALL_CAPS), 0); + + r = good_video(); + r.fourcc = XPSB_FOURCC_YV12; + r.nplanes = 3; + r.plane[0] = raw(XPSB_FMT_A8, 320, 240); + r.plane[1] = raw(XPSB_FMT_A8, 160, 120); + r.plane[2] = raw(XPSB_FMT_A8, 160, 120); + expect("YV12 with every capability", + (uint32_t)xpsb_3d_decide_video(&r, ALL_CAPS), 0); + + /* A missing conversion array is a wrong render, not a decline, so it + * has to be the decision that catches it. */ + r = good_video(); + r.conv = NULL; + expect("packed without conversion data", + (uint32_t)xpsb_3d_decide_video(&r, ALL_CAPS), 0); + + puts("video destination inferred from the box:"); + r = good_video(); + r.x = 60; + r.y = 80; + r.dst.w = 64; + r.dst.h = 48; + expect("box accepted", (uint32_t)xpsb_3d_video_dest(&d, &r), 0); + expect("width from the stride", d.w, 320); + expect("height covers the box", d.h, 128); + expect("decision follows", (uint32_t)xpsb_3d_decide_video(&r, ALL_CAPS), + 1); + + r.dst.stride = 4 * 300; + expect("stride not a whole 32-pixel width", + (uint32_t)xpsb_3d_video_dest(&d, &r), (uint32_t)-1); + expect("decision follows", (uint32_t)xpsb_3d_decide_video(&r, ALL_CAPS), + 0); + + r = good_video(); + r.dst.pict_format = PICT_a4r4g4b4; + expect("dest format with no surface code", + (uint32_t)xpsb_3d_decide_video(&r, ALL_CAPS), 0); + + r = good_video(); + expect("without the video capability", + (uint32_t)xpsb_3d_decide_video(&r, ALL_CAPS & ~XPSB_3D_CAP_VIDEO), + 0); + expect("with no capabilities", (uint32_t)xpsb_3d_decide_video(&r, 0), 0); +} + +/* What the module does as it stands, rather than what it would do with a + * complete frame layer: XPSB_3D_FRAME_CAPS carries the scalar and video bits, + * so the two one-sampled-unit composites and a packed YUV blit go to the SGX; + * the mask-texture bit is clear, so two-texture composites and every planar + * FOURCC still go to the CPU, and XPSB_NO_3D forces that everywhere. */ +static void test_shipped_caps(void) +{ + struct xpsb_3d_comp c = good_comp(), g = good_comp(), m = good_comp(); + struct xpsb_3d_video v = good_video(), p = good_video(); + unsigned caps; + + p.fourcc = XPSB_FOURCC_NV12; + p.nplanes = 2; + p.plane[0] = raw(XPSB_FMT_A8, 320, 240); + p.plane[1] = raw(XPSB_FMT_A8, 160, 120); + p.conv = NULL; + + g.scalar_src = 1; + g.scalar_mask = 0; + g.have_src = 0; + g.mask = surf(PICT_a8, 64, 64); + g.have_mask = 1; + + m.scalar_mask = 0; + m.mask = surf(PICT_a8, 64, 64); + m.have_mask = 1; + + puts("as built:"); + unsetenv("XPSB_NO_3D"); + caps = xpsb_3d_caps(); + printf(" capabilities %#x%s\n", caps, + (caps & XPSB_3D_CAP_COMPOSITE) ? "" : ", no composite shader"); + expect("quad capability present", caps & XPSB_3D_CAP_QUAD, + XPSB_3D_CAP_QUAD); + expect("scalar capability present", caps & XPSB_3D_CAP_SCALAR, + XPSB_3D_CAP_SCALAR); + expect("mask-texture capability absent", caps & XPSB_3D_CAP_MASK_TEX, 0); + expect("video capability present", caps & XPSB_3D_CAP_VIDEO, + XPSB_3D_CAP_VIDEO); + expect("scalar mask composite decision", + (uint32_t)xpsb_3d_decide_composite(&c, caps), 1); + expect("scalar source composite decision", + (uint32_t)xpsb_3d_decide_composite(&g, caps), 1); + expect("two-texture composite decision", + (uint32_t)xpsb_3d_decide_composite(&m, caps), 0); + expect("packed video decision", (uint32_t)xpsb_3d_decide_video(&v, caps), + 1); + expect("planar video decision", (uint32_t)xpsb_3d_decide_video(&p, caps), + 0); + + /* what raising the mask-texture bit would change, and nothing else: the + * two-texture composite goes to the SGX, the two scalar forms stay + * where they are, and the planar FOURCCs still decline on their own */ + caps |= XPSB_3D_CAP_MASK_TEX; + expect("with the bit, two textures", + (uint32_t)xpsb_3d_decide_composite(&m, caps), 1); + expect("with the bit, scalar mask unchanged", + (uint32_t)xpsb_3d_decide_composite(&c, caps), 1); + expect("with the bit, scalar source unchanged", + (uint32_t)xpsb_3d_decide_composite(&g, caps), 1); + expect("with the bit, planar video unchanged", + (uint32_t)xpsb_3d_decide_video(&p, caps), 0); + caps = xpsb_3d_caps(); + + setenv("XPSB_NO_3D", "1", 1); + expect("XPSB_NO_3D=1", xpsb_3d_caps(), 0); + setenv("XPSB_NO_3D", "yes", 1); + expect("XPSB_NO_3D=yes", xpsb_3d_caps(), 0); + setenv("XPSB_NO_3D", "0", 1); + expect("XPSB_NO_3D=0", xpsb_3d_caps(), caps); + setenv("XPSB_NO_3D", "", 1); + expect("XPSB_NO_3D empty", xpsb_3d_caps(), caps); + unsetenv("XPSB_NO_3D"); + + expect("composite with 3D off", + (uint32_t)xpsb_3d_decide_composite(&c, 0), 0); + expect("video with 3D off", (uint32_t)xpsb_3d_decide_video(&v, 0), 0); +} + +int main(void) +{ + test_format(); + test_surfaces(); + test_composite(); + test_scalar_operands(); + test_video(); + test_shipped_caps(); + + printf("\n%s: %d failure%s\n", failures ? "FAILED" : "ok", failures, + failures == 1 ? "" : "s"); + return failures != 0; +} diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/test_frame.c mesa-26.2.2/src/gallium/drivers/sgx/test_frame.c --- mesa-26.2.2.orig/src/gallium/drivers/sgx/test_frame.c 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/test_frame.c 2026-09-08 10:57:36.684534679 +0200 @@ -0,0 +1,2455 @@ +/* Self-test for the extracted SGX frame renderer. + * + * test/drmcube.c is compiled into this program with its main() renamed, so + * its generators are callable side by side with the extracted ones. Every + * buffer a frame carries is generated by both and required to be identical + * dword for dword; the geometry parameterisation is then checked by requiring + * that a non-square target differs from drmcube's square-only output in + * exactly the six heap floats that carry the height. + * + * The generalisations on top of that are checked the same way: the regenerated + * relocation tables must be byte-identical to test/reloc_tables.h before any + * other configuration is believed, the caller-supplied destination must be a + * no-op on the surface gen_heap already describes, and the quad-only draw and + * rastgeom arrays must carry the same records drmcube emits, moved. + * + * Needs neither hardware nor DRM: the generators are pure functions of the + * geometry. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +/* The reference is a whole program, and it must come before any system header + * so that its own feature-test macros take effect. Only its generators are + * wanted here; the scaffolding it does not call must not fail the build. */ +#pragma GCC diagnostic push +#pragma GCC diagnostic ignored "-Wunused-function" +#pragma GCC diagnostic ignored "-Wunused-but-set-variable" +#define main drmcube_main +#include "../../test/drmcube.c" +#undef main +#pragma GCC diagnostic pop + +#include +#include +#include + +#include "xpsb_frame.h" +#include "xpsb_pds.h" + +#include "xpsb_yuv.h" + +/* xpsb_frame.c calls xpsb_composite_shader(), and the Makefile's test_frame + * rule links only test_frame.c, xpsb_frame.c and xpsb_vidshader.c. */ +#include "xpsb_shader.c" + +/* The state block's dwords through its layout rather than by number: the + * block moves, so a fixed index would test the capture and not the code. */ +#define PDS_SEC(h) xpsb_heap_pds_dw((h), XPSB_PDS_W_SEC) +#define PDS_CTL(h) xpsb_heap_pds_dw((h), XPSB_PDS_W_CTL) +#define PDS_PRI(h) xpsb_heap_pds_dw((h), XPSB_PDS_W_PRI) +#define STATE(h, g) ((unsigned)xpsb_heap_state_off((h), (g))) +#define ISP_A(h) STATE((h), XPSB_STATE_ISP_A) + +#define HEAP_DW 0x11000 +#define RAST_DW 0x800 +#define MISC_DW 64 + +static int failures; + +static void expect(const char *what, uint32_t got, uint32_t want) +{ + if (got != want) { + printf(" FAIL %-32s got %#010x want %#010x\n", what, got, want); + failures++; + } +} + +/* Report the first few differing dwords rather than just the count: a stray + * field is only actionable if its offset is named. */ +static void expect_same(const char *what, const uint32_t *a, const uint32_t *b, + unsigned n) +{ + unsigned i, bad = 0; + + for (i = 0; i < n; i++) { + if (a[i] == b[i]) continue; + if (bad < 8) + printf(" FAIL %-24s dword %#06x: %#010x != %#010x\n", + what, i, a[i], b[i]); + bad++; + } + if (bad) { + printf(" FAIL %-24s %u dword(s) differ\n", what, bad); + failures++; + } +} + +static void test_heap(void) +{ + static const int sizes[] = { 16, 64, 128, 144, 256, 512, 640 }; + uint32_t *mine = calloc(HEAP_DW, 4), *ref = calloc(HEAP_DW, 4); + unsigned i, d; + + puts("parameter heap, against drmcube's gen_heap:"); + for (i = 0; i < sizeof sizes / sizeof sizes[0]; i++) + for (d = 0; d < 2; d++) { + char what[64]; + int s = sizes[i]; + + memset(mine, 0, HEAP_DW * 4); + memset(ref, 0, HEAP_DW * 4); + xpsb_gen_heap(mine, s, s, 0, (int)d); + gen_heap(ref, s, s, (int)d); + snprintf(what, sizeof what, "heap %dx%d depth=%u", s, s, d); + expect_same(what, mine, ref, HEAP_DW); + printf(" %-24s stride %u, 0x000=%#010x 0x003=%#010x\n", + what, ((unsigned)s + 31u) & ~31u, + mine[0x000], mine[0x003]); + } + free(mine); + free(ref); +} + +/* drmcube feeds all twelve size floats (float)W, so a non-square target must + * differ from it in exactly the six that carry the height. */ +static void test_heap_nonsquare(void) +{ + static const unsigned hf[6] = { + 0x078, 0x07c, 0x1102, 0x1106, 0x1112, 0x1116 + }; + uint32_t *mine = calloc(HEAP_DW, 4), *ref = calloc(HEAP_DW, 4); + float fh = 128.0f, fw = 256.0f; + uint32_t want_h, want_w; + unsigned i, j, extra = 0; + + memcpy(&want_h, &fh, 4); + memcpy(&want_w, &fw, 4); + xpsb_gen_heap(mine, 256, 128, 0, 1); + gen_heap(ref, 256, 128, 1); + + puts("non-square 256x128:"); + for (i = 0; i < HEAP_DW; i++) { + if (mine[i] == ref[i]) continue; + for (j = 0; j < 6 && hf[j] != i; j++) + ; + if (j == 6) { + printf(" FAIL unexpected difference at %#06x: " + "%#010x != %#010x\n", i, mine[i], ref[i]); + extra++; + } + } + if (extra) failures++; + for (i = 0; i < 6; i++) { + char what[32]; + + snprintf(what, sizeof what, "height float %#05x", hf[i]); + expect(what, mine[hf[i]], want_h); + snprintf(what, sizeof what, "drmcube fed width at %#05x", hf[i]); + expect(what, ref[hf[i]], want_w); + } + expect("0x003 height|stride", mine[0x003], (127u << 12) | 255u); + expect("0x04b width|height", mine[0x04b], 0x6c000000u | (255u << 12) | 127u); + expect("0x05a tile shadow", mine[0x05a], (16u << 16) | 8u); + free(mine); + free(ref); +} + +/* The destination stride is a free field in the two render-target surface + * descriptors, so it must reach 0x000 and 0x003 without disturbing the rest. */ +static void test_heap_stride(void) +{ + uint32_t *a = calloc(HEAP_DW, 4), *b = calloc(HEAP_DW, 4); + unsigned i, diff = 0; + + xpsb_gen_heap(a, 200, 100, 0, 1); + xpsb_gen_heap(b, 200, 100, 320, 1); + + puts("explicit destination stride:"); + expect("default 0x000", a[0x000], 223u << 15); + expect("default 0x003", a[0x003], (99u << 12) | 223u); + expect("stride 320 0x000", b[0x000], 319u << 15); + expect("stride 320 0x003", b[0x003], (99u << 12) | 319u); + for (i = 0; i < HEAP_DW; i++) + if (a[i] != b[i] && i != 0x000 && i != 0x003) diff++; + expect("only 0x000/0x003 move", diff, 0); + free(a); + free(b); +} + +static void test_rastgeom(void) +{ + static const int sizes[] = { 16, 128, 256, 640 }; + uint32_t *mine = calloc(RAST_DW, 4), *ref = calloc(RAST_DW, 4); + unsigned i; + + /* Byte for byte against the capture, with no licensed deviation. The + * covering triangle that used to sit here was measured inert on the + * part and reverted; SGX_BG_SCALE is what moves the record now, and + * unset it must reproduce drmcube exactly - which is what makes the + * sweep's 1.0 a control. */ + puts("rastgeom, against drmcube's gen_rastgeom:"); + for (i = 0; i < sizeof sizes / sizeof sizes[0]; i++) { + char what[32]; + + memset(mine, 0, RAST_DW * 4); + memset(ref, 0, RAST_DW * 4); + xpsb_gen_rastgeom(mine, sizes[i], sizes[i]); + gen_rastgeom(ref, sizes[i], sizes[i]); + snprintf(what, sizeof what, "rastgeom %d", sizes[i]); + expect_same(what, mine, ref, RAST_DW); + } + memset(mine, 0, RAST_DW * 4); + memset(ref, 0, RAST_DW * 4); + xpsb_gen_rastgeom(mine, 256, 128); + gen_rastgeom(ref, 256, 128); + expect_same("rastgeom 256x128", mine, ref, RAST_DW); + printf(" rastgeom present blit at (%d,%d), 4 records checked\n", + XPSB_PRESENT_X, XPSB_PRESENT_Y); + free(mine); + free(ref); +} + +static void test_streams(void) +{ + static const int sizes[] = { 16, 128, 256, 512, 640 }; + uint32_t mine[MISC_DW], ref[MISC_DW]; + unsigned i, n, m; + + puts("draw records and register streams:"); + memset(mine, 0, sizeof mine); + memset(ref, 0, sizeof ref); + xpsb_gen_draw_records(mine); + gen_draw_records(ref); + expect_same("draw records", mine, ref, MISC_DW); + expect("draw command offset", mine[XPSB_DRAW_CMD_OFF / 4] & 0xfff00000u, + XPSB_DRAW_CMD_TAG); + + memset(mine, 0, sizeof mine); + memset(ref, 0, sizeof ref); + n = xpsb_gen_ta_stream(mine); + m = gen_ta_stream(ref, 128, 128); + expect("ta stream dwords", n, m); + expect_same("ta stream", mine, ref, MISC_DW); + + memset(mine, 0, sizeof mine); + memset(ref, 0, sizeof ref); + n = xpsb_gen_oom_stream(mine); + m = gen_oom_stream(ref); + expect("oom stream dwords", n, m); + /* One deliberate deviation from the capture: bit 8 of the 0x4bc + * background word. The captured stream has 0x200 and this frame needs + * 0x300, or every out-of-memory recovery loses the tail of whatever + * macro tile it flushed - measured across three scenes and three heap + * sizes. The reference stays at the captured value so the deviation + * has to be stated here rather than quietly rebaselined. */ + expect("oom stream 0x4bc, deviating from the capture", mine[3], 0x300); + ref[3] = mine[3]; + expect_same("oom stream, the rest", mine, ref, MISC_DW); + + for (i = 0; i < sizeof sizes / sizeof sizes[0]; i++) { + char what[40]; + int s = sizes[i]; + + memset(mine, 0, sizeof mine); + memset(ref, 0, sizeof ref); + n = xpsb_gen_ta_raster_stream(mine, s, s); + m = gen_ta_raster_stream(ref, s, s); + snprintf(what, sizeof what, "ta-raster stream %d", s); + expect("ta-raster dwords", n, m); + /* The second deliberate deviation: register 0x0410 is + * EUR_CR_ISP_RENDBOX2, the last tile of the render box and not + * how many tiles there are, so it is one less than drmcube + * writes. The vendor's own formula is ceil(x1 / 16) - 1 + * (SGXTQ_SetupTransferRenderBox, sgxtransfer_utils.c:3259) and + * the present chunk below already writes an inclusive last + * tile from a capture. Every size swept here divides by + * sixteen, which is exactly where the two disagree. */ + expect(what, mine[7], + ((uint32_t)(s / 16 - 1) << 16) | (uint32_t)(s / 16 - 1)); + ref[7] = mine[7]; + expect_same(what, mine, ref, MISC_DW); + + memset(mine, 0, sizeof mine); + memset(ref, 0, sizeof ref); + n = xpsb_gen_raster_stream(mine, s, s); + m = gen_raster_stream(ref, s, s); + snprintf(what, sizeof what, "raster stream %d", s); + expect("raster dwords", n, m); + expect_same(what, mine, ref, MISC_DW); + } + memset(mine, 0, sizeof mine); + memset(ref, 0, sizeof ref); + xpsb_gen_ta_raster_stream(mine, 256, 128); + gen_ta_raster_stream(ref, 256, 128); + expect("ta-raster 256x128 render box", mine[7], (15u << 16) | 7u); + ref[7] = mine[7]; + expect_same("ta-raster 256x128", mine, ref, MISC_DW); + memset(mine, 0, sizeof mine); + memset(ref, 0, sizeof ref); + xpsb_gen_raster_stream(mine, 256, 128); + gen_raster_stream(ref, 256, 128); + expect_same("raster 256x128", mine, ref, MISC_DW); +} + +/* Multisampling: the two reachable sample counts, and that everything else is + * refused rather than programmed as something. */ +static void test_msaa(void) +{ + uint32_t ta[MISC_DW], ras[MISC_DW]; + unsigned n, m, i; + uint32_t box1 = 0, box4 = 0, ext1 = 0, ext4 = 0; + + puts("multisample:"); + expect("axis at 1 sample", xpsb_msaa_axis(XPSB_MSAA_1X), 1); + expect("axis at 4 samples", xpsb_msaa_axis(XPSB_MSAA_4X), 2); + expect("2 samples is not reachable", xpsb_msaa_axis(2), 0); + expect("8 samples is not reachable", xpsb_msaa_axis(8), 0); + expect("0 samples is not reachable", xpsb_msaa_axis(0), 0); + /* The pixel centre in sixteenths, and the rotated grid. */ + expect("1x positions", xpsb_msaa_positions(XPSB_MSAA_1X), 0x00000088u); + expect("4x positions", xpsb_msaa_positions(XPSB_MSAA_4X), 0xeaa26e26u); + expect("no positions for 2", xpsb_msaa_positions(2), 0); + + memset(ta, 0, sizeof ta); + n = xpsb_gen_ta_stream(ta); + expect("the captured TA stream is single sample", + (uint32_t)xpsb_ta_set_msaa(ta, n, XPSB_MSAA_1X), 0); + for (i = 0; i + 1 < n; i += 2) { + if (ta[i] == 0x0204) + expect("TE_AA off at one sample", ta[i + 1], 0); + if (ta[i] == 0x0250) + expect("MTE positions at one sample", ta[i + 1], + 0x00000088u); + } + expect("4x TA accepted", + (uint32_t)xpsb_ta_set_msaa(ta, n, XPSB_MSAA_4X), 0); + for (i = 0; i + 1 < n; i += 2) { + if (ta[i] == 0x0204) + expect("TE_AA x and y at 2x2", ta[i + 1], 0xc0000000u); + if (ta[i] == 0x0250) + expect("MTE positions at 2x2", ta[i + 1], 0xeaa26e26u); + } + expect("a TA stream is not touched at 2 samples", + (uint32_t)xpsb_ta_set_msaa(ta, n, 2), (uint32_t)-1); + for (i = 0; i + 1 < n; i += 2) + if (ta[i] == 0x0204) + expect("and keeps what it had", ta[i + 1], 0xc0000000u); + + memset(ras, 0, sizeof ras); + m = xpsb_gen_ta_raster_stream(ras, 256, 256); + for (i = 0; i + 1 < m; i += 2) { + if (ras[i] == 0x0410) + box1 = ras[i + 1]; + if (ras[i] == 0x0480) + ext1 = (ras[i + 1] >> 8) & 0xffu; + } + expect("4x raster accepted", + (uint32_t)xpsb_ta_raster_set_msaa(ras, m, 256, 256, + XPSB_MSAA_4X), 0); + for (i = 0; i + 1 < m; i += 2) { + if (ras[i] == 0x04c8) + expect("ISP positions at 2x2", ras[i + 1], 0xeaa26e26u); + if (ras[i] == 0x042c) + expect("AA mode enabled", ras[i + 1], 1); + if (ras[i] == 0x0410) + box4 = ras[i + 1]; + if (ras[i] == 0x0480) + ext4 = (ras[i + 1] >> 8) & 0xffu; + } + /* The field is the last tile inclusive, so the sample-tile count is + * what doubles: (last + 1) * 2 - 1, not last * 2. */ + expect("render box x doubles", (box4 >> 16) & 0xffu, + (((box1 >> 16) & 0xffu) + 1u) * 2u - 1u); + expect("render box y doubles", box4 & 0xffu, + ((box1 & 0xffu) + 1u) * 2u - 1u); + expect("ZLS extent doubles", ext4 + 1u, (ext1 + 1u) * 2u); + expect("back to one sample", + (uint32_t)xpsb_ta_raster_set_msaa(ras, m, 256, 256, + XPSB_MSAA_1X), 0); + for (i = 0; i + 1 < m; i += 2) { + if (ras[i] == 0x042c) + expect("AA mode off again", ras[i + 1], 0); + if (ras[i] == 0x0410) + expect("render box back", ras[i + 1], box1); + if (ras[i] == 0x0480) + expect("ZLS extent back", (ras[i + 1] >> 8) & 0xffu, + ext1); + } + /* The field holds the last tile, so its 255 is 256 sample tiles - the + * whole 2048 the part allows at one sample, reached exactly at 2x2. + * Multisampling costs no render size here; a count would have lost the + * last row and column to the field. */ + memset(ras, 0, sizeof ras); + m = xpsb_gen_ta_raster_stream(ras, 2048, 2048); + expect("2048 square at one sample", + (uint32_t)xpsb_ta_raster_set_msaa(ras, m, 2048, 2048, + XPSB_MSAA_1X), 0); + expect("2048 square fits at 2x2", + (uint32_t)xpsb_ta_raster_set_msaa(ras, m, 2048, 2048, + XPSB_MSAA_4X), 0); + expect("2064 square is refused at 2x2", + (uint32_t)xpsb_ta_raster_set_msaa(ras, m, 2064, 2064, + XPSB_MSAA_4X), (uint32_t)-1); +} + +/* The relocation lists and the buffer table are carried literals, so the copy + * has to be checked against test/reloc_tables.h and test/frame.h themselves. */ +static void test_tables(void) +{ + unsigned i; + + puts("carried tables, against test/reloc_tables.h and test/frame.h:"); + expect("ta reloc count", sizeof ta_relocs / sizeof ta_relocs[0], + XPSB_NUM_TA_RELOCS); + expect("raster reloc count", + sizeof raster_relocs / sizeof raster_relocs[0], XPSB_NUM_RAS_RELOCS); + expect("ta reloc bytes", sizeof ta_relocs, sizeof xpsb_ta_relocs); + expect("raster reloc bytes", sizeof raster_relocs, sizeof xpsb_raster_relocs); + expect("ta relocs", memcmp(xpsb_ta_relocs, ta_relocs, + sizeof xpsb_ta_relocs) != 0, 0); + expect("raster relocs", memcmp(xpsb_raster_relocs, raster_relocs, + sizeof xpsb_raster_relocs) != 0, 0); + expect("buffer count", FRAME_NBUF, XPSB_FRAME_NBUF); + for (i = 0; i < XPSB_FRAME_NBUF; i++) { + char what[32]; + + snprintf(what, sizeof what, "buffer %u size", i); + expect(what, (uint32_t)xpsb_frame_bufs[i].size, + (uint32_t)frame_bufs[i].size); + snprintf(what, sizeof what, "buffer %u create", i); + expect(what, (uint32_t)xpsb_frame_bufs[i].create, + (uint32_t)frame_bufs[i].create); + expect("create high", (uint32_t)(xpsb_frame_bufs[i].create >> 32), + (uint32_t)(frame_bufs[i].create >> 32)); + snprintf(what, sizeof what, "buffer %u vflags", i); + expect(what, (uint32_t)xpsb_frame_bufs[i].vflags, + (uint32_t)frame_bufs[i].vflags); + snprintf(what, sizeof what, "buffer %u vmask", i); + expect(what, (uint32_t)xpsb_frame_bufs[i].vmask, + (uint32_t)frame_bufs[i].vmask); + expect("ta list", (uint32_t)xpsb_ta_list[i], (uint32_t)ta_list[i]); + } + for (i = 0; i < XPSB_RAS_LIST_LEN; i++) { + expect("raster list", (uint32_t)xpsb_raster_list[i], + (uint32_t)raster_list[i]); + expect("raster vflags", (uint32_t)xpsb_raster_vflags[i], + (uint32_t)raster_vflags[i]); + expect("raster vmask", (uint32_t)xpsb_raster_vmask[i], + (uint32_t)raster_vmask[i]); + } + expect("scene num_buffers", XPSB_SCENE_NB, SCENE_NB); + expect("ta engine", XPSB_TA_ENGINE, TA_ENGINE); + expect("ta flags", XPSB_TA_TAFLAGS, TA_TAFLAGS); + expect("ta cmd offset", XPSB_TA_CMD_OFF, TA_CMD_OFF); + expect("ta stream offset", XPSB_TA_TA_OFF, TA_TA_OFF); + expect("oom offset", XPSB_TA_OOM_OFF, TA_OOM_OFF); + expect("raster engine", XPSB_RAS_ENGINE, RAS_ENGINE); + expect("raster cmd offset", XPSB_RAS_CMD_OFF, RAS_CMD_OFF); + expect("raster ta size", XPSB_RAS_TA_SIZE, RAS_TA_SIZE); + expect("raster oom size", XPSB_RAS_OOM_SIZE, RAS_OOM_SIZE); + expect("vertex offset", XPSB_VTX_OFF, VTX_OFF); + expect("vertex stride", XPSB_VTX_STRIDE, VTX_STRIDE); + expect("index offset", XPSB_IDX_OFF, IDX2_OFF); + expect("draw command offset", XPSB_DRAW_CMD_OFF, DRAW_CMD_OFF); + expect("texture control offset", XPSB_TEXCTL_OFF, TEXCTL_OFF); + expect("texture state offset", XPSB_TEXSTATE_OFF, TEXSTATE_OFF); + expect("present x", XPSB_PRESENT_X, PRESENT_X); + expect("present y", XPSB_PRESENT_Y, PRESENT_Y); + /* At the reference's own eleven-float record: XPSB_MAX_VTX itself now + * assumes the widest record the frame can describe, so that the bound + * holds whatever stride a caller writes. */ + expect("max vertices", XPSB_MAX_VTX_AT(XPSB_VTX_STRIDE), MAX_VTX); + expect("texture bytes", XPSB_TEX_BYTES, ATLAS_BYTES); +} + +/* disasm/emit-pixel-shader.md sections 3.1-3.5. The 64x16 a8r8g8b8 word is the + * one value two independent decodes agree on, so it is the anchor. */ +static void test_mipmap_state(void); +static void test_volume_state(void); + +static void test_texstate(void) +{ + struct xpsb_surface_desc s; + uint32_t w0, w1; + unsigned i; + + puts("texture state words:"); + memset(&s, 0, sizeof s); + s.w = 64; s.h = 16; s.format = XPSB_FMT_8888; + s.stride = 4 * 64; + expect("64x16 8888 accepted", xpsb_surface_check(&s) != 0, 0); + expect("tex_state ok", xpsb_tex_state(&w0, &w1, &s) != 0, 0); + expect("word1 64x16 8888", w1, 0x6c03f00fu); + expect("word1 arithmetic", w1, + 0x60000000u | (0x0cu << 24) | (63u << 12) | 15u); + expect("word0 repeat/nearest", w0, 0x03fe0000u); + + s.umode = XPSB_WRAP_CLAMP; s.vmode = XPSB_WRAP_CLAMP; + expect("tex_state clamp", xpsb_tex_state(&w0, &w1, &s) != 0, 0); + expect("word0 clamp/clamp", w0, 0x03fe0090u); + s.minfilter = XPSB_FILTER_LINEAR; s.magfilter = XPSB_FILTER_LINEAR; + expect("tex_state linear", xpsb_tex_state(&w0, &w1, &s) != 0, 0); + expect("word0 clamp+linear", w0, 0x03fe1490u); + s.umode = XPSB_WRAP_CLAMPGL; s.vmode = XPSB_WRAP_CLAMPGL; + expect("tex_state clampGL", xpsb_tex_state(&w0, &w1, &s) != 0, 0); + expect("word0 clampGL+linear", w0, 0x03fe15f8u); + + memset(&s, 0, sizeof s); + s.w = 64; s.h = 16; s.format = XPSB_FMT_A8; s.stride = 64; + expect("tex_state a8", xpsb_tex_state(&w0, &w1, &s) != 0, 0); + expect("word0 8-bit bit 30", w0, 0x43fe0000u); + expect("word1 a8 64x16", w1, 0x6003f00fu); + + memset(&s, 0, sizeof s); + s.w = 64; s.h = 16; s.format = XPSB_FMT_565; s.stride = 2 * 64; + expect("tex_state 565", xpsb_tex_state(&w0, &w1, &s) != 0, 0); + expect("word1 565", w1, 0x6503f00fu); + expect("word0 565 no bit 30", w0, 0x03fe0000u); + + expect("bpp a8", xpsb_format_bpp(XPSB_FMT_A8), 1); + expect("bpp 1555", xpsb_format_bpp(XPSB_FMT_1555), 2); + expect("bpp yuy2", xpsb_format_bpp(XPSB_FMT_YUY2), 2); + expect("bpp uyvy", xpsb_format_bpp(XPSB_FMT_UYVY), 2); + expect("bpp 8888", xpsb_format_bpp(XPSB_FMT_8888), 4); + expect("bpp bgr8888", xpsb_format_bpp(XPSB_FMT_BGR8888), 4); + expect("bpp unknown", xpsb_format_bpp(0x1f), 0); + + /* The two drivers disagree on word 0 bits 25:21 and bit 1. The capture + * is the default because it is the dword drmcube already writes into + * heap+0x348 in the stream that renders; Xpsb's is a mode. Both are + * asserted so neither can move silently, and the default is held + * against drmcube's own tex_ctl() over the whole filter/wrap matrix. */ + memset(&s, 0, sizeof s); + s.w = 64; s.h = 64; s.format = XPSB_FMT_8888; s.stride = 4 * 64; + s.umode = XPSB_WRAP_CLAMP; s.vmode = XPSB_WRAP_CLAMP; + expect("tex_state default mode", xpsb_tex_state(&w0, &w1, &s) != 0, 0); + expect("drmcube's captured word1", w1, 0x6c03f03fu); + expect("default is the captured word0", w0, 0x03fe0090u); + expect("tex_state XPSB mode", + xpsb_tex_state_mode(&w0, &w1, &s, XPSB_TEXCTL_XPSB) != 0, 0); + expect("XPSB mode word0", w0, 0x001e0092u); + expect("the two differ in bits 25:21 and 1", + w0 ^ 0x03fe0090u, 0x03e00002u); + expect("unknown mode refused", + xpsb_tex_state_mode(&w0, &w1, &s, 2) == 0, 0); + + for (i = 0; i < 4; i++) { + int linear = i & 1, repeat = (i >> 1) & 1; + char what[48]; + + s.umode = s.vmode = repeat ? XPSB_WRAP_REPEAT : XPSB_WRAP_CLAMP; + s.minfilter = s.magfilter = linear ? XPSB_FILTER_LINEAR + : XPSB_FILTER_NEAREST; + xpsb_tex_state(&w0, &w1, &s); + snprintf(what, sizeof what, "tex_ctl(%d,%d)", linear, repeat); + expect(what, w0, tex_ctl(linear, repeat)); + snprintf(what, sizeof what, "tex_state(64,64,%d)", 0); + expect(what, w1, tex_state(64, 64, 0)); + } + printf(" 64x16 a8r8g8b8: word0 %#010x (XPSB mode %#010x) word1 %#010x\n", + 0x03fe0000u, 0x001e0002u, 0x6c03f00fu); + + test_mipmap_state(); +} + +/* gl-re/textures.md section 3.1 and 5.1: the level count in word 0 bits 20:17, + * the trilinear bit, the twiddled word 1 form, and the Morton order itself. */ +static void test_mipmap_state(void) +{ + struct xpsb_surface_desc s; + uint32_t w0, w1; + unsigned x, y; + + puts("mipmap and twiddled state:"); + memset(&s, 0, sizeof s); + s.w = 64; s.h = 64; s.format = XPSB_FMT_8888; s.stride = 4 * 64; + + expect("no chain keeps the saturated field", + (xpsb_tex_state(&w0, &w1, &s), w0 & 0x001e0000u), 0x001e0000u); + s.nlevels = 1; + expect("one level is still not mipmapped", + (xpsb_tex_state(&w0, &w1, &s), w0 & 0x001e0000u), 0x001e0000u); + s.nlevels = 7; + expect("seven levels of 64x64", xpsb_tex_state(&w0, &w1, &s) != 0, 0); + expect("top level index is 6", (w0 >> 17) & 0xfu, 6u); + expect("mip filter nearest leaves bit 9", w0 & 0x200u, 0u); + s.mipfilter = XPSB_MIPFILTER_LINEAR; + xpsb_tex_state(&w0, &w1, &s); + expect("trilinear sets bit 9", w0 & 0x200u, 0x200u); + expect("and is orthogonal to the other two", + w0 & 0x00003c00u, 0u); + s.nlevels = 8; + expect("more levels than 64x64 has refused", + xpsb_tex_state(&w0, &w1, &s) == 0, 0); + + memset(&s, 0, sizeof s); + s.w = 64; s.h = 16; s.format = XPSB_FMT_8888; s.twiddled = 1; + expect("twiddled needs no stride", xpsb_tex_state(&w0, &w1, &s) != 0, 0); + expect("twiddled word1 is log2 sizes", w1, + ((uint32_t)XPSB_FMT_8888 << 24) | (6u << 16) | 4u); + expect("twiddled class is 0b000", w1 & 0xe0000000u, + XPSB_SURF_2D_TWIDDLED); + s.w = 60; + expect("twiddled refuses non-power-of-two", + xpsb_tex_state(&w0, &w1, &s) == 0, 0); + + /* Square: pure Morton, Y in the even bit of each pair. */ + expect("twiddle (0,0)", xpsb_twiddle_index(0, 0, 3, 3), 0u); + expect("twiddle (0,1)", xpsb_twiddle_index(0, 1, 3, 3), 1u); + expect("twiddle (1,0)", xpsb_twiddle_index(1, 0, 3, 3), 2u); + expect("twiddle (1,1)", xpsb_twiddle_index(1, 1, 3, 3), 3u); + expect("twiddle (0,2)", xpsb_twiddle_index(0, 2, 3, 3), 4u); + expect("twiddle (2,0)", xpsb_twiddle_index(2, 0, 3, 3), 8u); + expect("twiddle (7,7)", xpsb_twiddle_index(7, 7, 3, 3), 63u); + /* Wider than tall: the extra X bits sit above the square block. */ + expect("twiddle wide (4,0)", xpsb_twiddle_index(4, 0, 3, 2), 16u); + expect("twiddle wide (3,3)", xpsb_twiddle_index(3, 3, 3, 2), 15u); + /* Taller than wide, the mirror case. */ + expect("twiddle tall (0,4)", xpsb_twiddle_index(0, 4, 2, 3), 16u); + + /* Every texel of a level must land on its own index, or the level + * overlaps itself and the test above would not notice. */ + { + unsigned char seen[64 * 16]; + int dup = 0; + + memset(seen, 0, sizeof seen); + for (y = 0; y < 16; y++) + for (x = 0; x < 64; x++) { + uint32_t i = xpsb_twiddle_index(x, y, 6, 4); + + if (i >= sizeof seen || seen[i]) dup = 1; + else seen[i] = 1; + } + expect("64x16 twiddle is a bijection", dup, 0); + } + + expect("level 0 is at zero", xpsb_twiddle_level_offset(64, 64, 0), 0u); + expect("level 1 follows level 0", + xpsb_twiddle_level_offset(64, 64, 1), 64u * 64u); + expect("levels are packed tight", + xpsb_twiddle_level_offset(64, 64, 2), 64u * 64u + 32u * 32u); + expect("the chain is 4/3 of the base", + xpsb_twiddle_level_offset(64, 64, 7), 5461u); + /* A rectangular chain stops shrinking the axis that reached one. */ + expect("rectangular chain", + xpsb_twiddle_level_offset(8, 2, 3), 16u + 4u + 2u); + + test_volume_state(); +} + +/* The volume, border and gamma fields, against sgxdefs.h's SGX535 branch: + * TEXTYPE_3D is 1 at DOUTT1[31:29], SSIZE is DOUTT1[15:12], SADDRMODE is + * DOUTT0[2:0], GAMMA is DOUTT0 bit 27, CHANREPLICATE bit 30, and the border + * codes are 5, 6 and 4. */ +static void test_volume_state(void) +{ + struct xpsb_surface_desc s; + uint32_t w0, w1, x, y, z; + + puts("volume, border and gamma state:"); + memset(&s, 0, sizeof s); + s.w = 8; s.h = 4; s.depth = 2; s.format = XPSB_FMT_8888; + expect("a linear volume has no encoding", + xpsb_surface_check(&s) == 0, 0); + s.twiddled = 1; + expect("a twiddled 8x4x2 is accepted", xpsb_surface_check(&s) != 0, 0); + expect("tex_state volume", xpsb_tex_state(&w0, &w1, &s) != 0, 0); + expect("word1 type 3D, log2 sizes, SSIZE", w1, + (1u << 29) | ((uint32_t)XPSB_FMT_8888 << 24) | (3u << 16) | + (1u << 12) | 2u); + expect("S repeat leaves [2:0] clear", w0 & 7u, 0u); + s.smode = XPSB_WRAP_CLAMP; + xpsb_tex_state(&w0, &w1, &s); + expect("S clamp is code 2 at [2:0]", w0 & 7u, 2u); + expect("and touches nothing else", w0 & ~7u, 0x03fe0000u); + s.depth = 3; + expect("a non-power-of-two depth is refused", + xpsb_surface_check(&s) == 0, 0); + s.depth = 8; + s.nlevels = 4; + expect("the chain may be as long as the depth allows", + xpsb_tex_state(&w0, &w1, &s) != 0, 0); + s.nlevels = 5; + expect("but no longer", xpsb_tex_state(&w0, &w1, &s) == 0, 0); + + /* The two twiddle orders, on a level where they differ. */ + expect("morton (0,0,0)", xpsb_twiddle3_index(0, 0, 0, 2, 2, 1, + XPSB_TWIDDLE3_MORTON), 0u); + expect("morton (0,1,0) is bit 0", + xpsb_twiddle3_index(0, 1, 0, 2, 2, 1, XPSB_TWIDDLE3_MORTON), 1u); + expect("morton (1,0,0) is bit 1", + xpsb_twiddle3_index(1, 0, 0, 2, 2, 1, XPSB_TWIDDLE3_MORTON), 2u); + expect("morton (0,0,1) is bit 2", + xpsb_twiddle3_index(0, 0, 1, 2, 2, 1, XPSB_TWIDDLE3_MORTON), 4u); + expect("morton (0,2,0) is bit 3 once z is out of bits", + xpsb_twiddle3_index(0, 2, 0, 2, 2, 1, XPSB_TWIDDLE3_MORTON), 8u); + expect("morton (2,0,0) is bit 4", + xpsb_twiddle3_index(2, 0, 0, 2, 2, 1, XPSB_TWIDDLE3_MORTON), 16u); + expect("slices (0,0,1) is a whole slice on", + xpsb_twiddle3_index(0, 0, 1, 2, 2, 1, XPSB_TWIDDLE3_SLICES), 16u); + expect("slices (2,0,0) is the 2D index", + xpsb_twiddle3_index(2, 0, 0, 2, 2, 1, XPSB_TWIDDLE3_SLICES), + xpsb_twiddle_index(2, 0, 2, 2)); + expect("a 2x2x2 is the same under both orders", + xpsb_twiddle3_index(1, 1, 1, 1, 1, 1, XPSB_TWIDDLE3_MORTON), + xpsb_twiddle3_index(1, 1, 1, 1, 1, 1, XPSB_TWIDDLE3_SLICES)); + expect("with no depth bits it is the 2D order", + xpsb_twiddle3_index(5, 3, 0, 3, 2, 0, XPSB_TWIDDLE3_MORTON), + xpsb_twiddle_index(5, 3, 3, 2)); + { + unsigned char seen[8 * 4 * 16]; + int dup = 0, o; + + for (o = 0; o < 2; o++) { + memset(seen, 0, sizeof seen); + for (z = 0; z < 16; z++) + for (y = 0; y < 4; y++) + for (x = 0; x < 8; x++) { + uint32_t i = xpsb_twiddle3_index( + x, y, z, 3, 2, 4, + (enum xpsb_twiddle3)o); + + if (i >= sizeof seen || seen[i]) + dup = 1; + else + seen[i] = 1; + } + expect(o ? "8x4x16 slices is a bijection" + : "8x4x16 morton is a bijection", dup, 0); + } + } + /* The issue a sampled set gets: TEXPROJ_RHW for every sample that is + * not a TXP - the vendor's one branch, usegles.c:2209-2219 - and + * TEXPROJ_T only for one that is, where the set carries its own + * divisor. */ + { + struct xpsb_attribs a; + + memset(&a, 0, sizeof a); + a.nset = 1; + a.set[0].width = 2; + a.set[0].sampled = 1; + expect("a sampled set is TEXPROJ_RHW", + xpsb_attrib_texproj(&a, 0), 1u); + expect("and two floats of record", xpsb_attrib_stride(&a), + 8u + 2u); + a.set[0].width = 4; + a.set[0].projected = 1; + expect("a projected sample is TEXPROJ_T", + xpsb_attrib_texproj(&a, 0), 2u); + a.set[0].projected = 0; + a.set[0].width = 3; + a.set[0].volume = 1; + expect("a volume set is TEXPROJ_RHW", + xpsb_attrib_texproj(&a, 0), 1u); + expect("and three floats of record", xpsb_attrib_stride(&a), + 8u + 3u); + /* Four wide is the shape the driver declares: the slice is + * then the third of four rather than the last of three, which + * is the component a sampled set's iterator has reason to + * take as a projector. The issue is RHW either way. */ + a.set[0].width = 4; + expect("a four-float volume set is TEXPROJ_RHW too", + xpsb_attrib_texproj(&a, 0), 1u); + expect("and four floats of record", xpsb_attrib_stride(&a), + 8u + 4u); + } + + /* Where the record must put the divisor. Only a projected sample has + * one in its own set: TEXPROJ_T divides by T and the declared + * dimension names it, the fourth float of a UVST set. Every other + * sample is TEXPROJ_RHW and divided by the position's plane, so no + * float of the set is a divisor and the record must not overwrite one + * with a 1.0. */ + { + struct xpsb_attribs a; + + memset(&a, 0, sizeof a); + a.nset = 2; + a.set[0].width = 2; + a.set[0].sampled = 1; + a.set[1].width = 2; + expect("an unprojected sample is divided by no float of its " + "own", (unsigned)(xpsb_attrib_proj_float(&a, 0) + 1), 0u); + expect("an iterated set is divided by nothing", + (unsigned)(xpsb_attrib_proj_float(&a, 1) + 1), 0u); + a.set[0].width = 4; + a.set[0].projected = 1; + expect("a projected UVST set divides by its fourth float", + (unsigned)(xpsb_attrib_proj_float(&a, 0) + 1), 4u); + a.nset = 1; + a.set[0].projected = 0; + a.set[0].width = 3; + a.set[0].volume = 1; + expect("a volume's third float is its slice, not a divisor", + (unsigned)(xpsb_attrib_proj_float(&a, 0) + 1), 0u); + } + + /* The MTE coordinate dimension is the set's own component count, as + * the vendor writes it: UV, UVS, UVST (opengles2/shader.c:2245-2270). + * Nothing declares UVT for an ordinary sample - T is a divisor, and a + * sample that needs one is projected. */ + { + struct xpsb_attribs a; + uint32_t *h = calloc(HEAP_DW, 4); + unsigned g14; + + if (!h) + return; + xpsb_gen_heap(h, 64, 64, 0, 0); + memset(&a, 0, sizeof a); + a.nset = 1; + a.set[0].width = 2; + a.set[0].sampled = 1; + expect("a two-float sampled set installs", + xpsb_heap_set_attribs(h, &a, NULL) == 0, 1u); + g14 = h[xpsb_heap_state_off(h, XPSB_STATE_TEXSIZE)]; + expect("and is declared UV", g14, 1u); + expect("and the record is two floats past the fixed eight", + h[XPSB_VTXSTRIDE_DW], (8u + 2u) * 4u); + a.set[0].width = 3; + expect("a three-float set installs", + xpsb_heap_set_attribs(h, &a, NULL) == 0, 1u); + expect("and is declared UVS", + h[xpsb_heap_state_off(h, XPSB_STATE_TEXSIZE)], 3u); + a.set[0].width = 4; + a.set[0].projected = 1; + expect("a projected set installs", + xpsb_heap_set_attribs(h, &a, NULL) == 0, 1u); + expect("and is declared UVST", + h[xpsb_heap_state_off(h, XPSB_STATE_TEXSIZE)], 7u); + free(h); + } + + /* The flat-shade corner reaches the iterator that carries the colour + * and nothing else: a texture coordinate is always interpolated + * (usegles.c:2240-2252). */ + { + struct xpsb_attribs a; + uint32_t *h = calloc(HEAP_DW, 4); + unsigned n = 0, i, flat = 0, gouraud = 0; + + if (!h) + return; + xpsb_gen_heap(h, 64, 64, 0, 0); + /* The shape a fixed-function draw takes under the routing: + * one iterated four-float set carrying the colour, and the + * TAG issue a textureless pixel task is given anyway. */ + memset(&a, 0, sizeof a); + a.nset = 1; + a.set[0].width = 4; + a.set[0].iterated = 1; + a.colour_set = 1; /* set 0 carries the colour */ + a.flatshade = 3; /* from vertex 2 */ + expect("a routed colour installs", + xpsb_heap_set_attribs(h, &a, NULL) == 0, 1u); + xpsb_attrib_issues(&a, &n, NULL); + expect("as the colour's issue and a TAG issue", n, 2u); + for (i = 0; i < n; i++) { + uint32_t w = h[XPSB_PRI_PDS_DW + + xpsb_pds_ds_dword(1, 1 + 2 * i)]; + unsigned use = (w >> 12) & 0xfu; + unsigned fs = (w >> XPSB_DOUTI_FLAT_SHIFT) & 3u; + + if (use == 0) + flat = fs + 1u; + else + gouraud += fs == 3u; + } + expect("the colour's issue flat-shades from vertex 2", flat, 3u); + expect("the TAG issue is left interpolated", gouraud, 1u); + a.flatshade = 0; + expect("and gouraud installs", xpsb_heap_set_attribs(h, &a, NULL) + == 0, 1u); + expect("with the colour interpolated again", + (h[XPSB_PRI_PDS_DW + xpsb_pds_ds_dword(1, 1)] >> + XPSB_DOUTI_FLAT_SHIFT) & 3u, 3u); + free(h); + } + + /* What the bases say and what the program does must agree. + * + * attrib_issues() appends a dummy TAG issue to a list that has none, + * because a textureless pixel task does not batch (e0477d9 measured + * 18x). It used to append it whenever n + 1 fit XPSB_PDS_MAX_ISSUE, + * which is four - but the generator cannot build four: issue i's + * texture address goes at ds1[2 + 2i] and ds1[8] is memory dword 24, + * past the sixteen the primary PDS region holds. So a three-set frame + * appended a fourth issue here, xpsb_heap_set_attribs() refused it + * there and shed it, and the two disagreed: the bases named a texel + * register the emitted program never produced. + * + * The bound asked is now the spare block's, not the captured slot's: + * xpsb_heap_set_attribs() builds against XPSB_PDS_MAX_DATA when the + * program goes there, so four issues are buildable, and + * xpsb_pri_pds_off() takes the spare block rather than shedding the + * tail into the captured slot. Both sides still describe the same + * frame; what changed is which frame that is. glmark2's jellyfish is + * three sets and was the shape that ended up with no texture issue at + * all - 1 frame a second against 5 with the tail kept. */ + { + struct xpsb_attribs a; + unsigned c = 0, sb[XPSB_NSET_MAX] = { 0 }; + unsigned tb[XPSB_MAX_TEX] = { 0 }, n = 0, i; + + expect("three issues fit the primary program", + (unsigned)xpsb_pds_issues_fit(3) != 0u, 1u); + expect("four do not", (unsigned)xpsb_pds_issues_fit(4), 0u); + + memset(&a, 0, sizeof a); + a.nset = 3; + for (i = 0; i < 3; i++) { + a.set[i].width = 4; + a.set[i].iterated = 1; + } + expect("three iterated sets get a texture issue", + xpsb_attrib_issues(&a, &n, NULL), 1u); + expect("and four issues, the tail last", n, 4u); + xpsb_attrib_bases(&a, &c, sb, tb); + expect("set 0 at register 0", sb[0], 0u); + expect("set 1 behind it", sb[1], 4u); + expect("set 2 behind that", sb[2], 8u); + /* The tail follows all three rather than sitting between two, + * so the texel it produces lands past the last set. */ + expect("and the texel register follows them", tb[0], 12u); + + /* Two sets leave room for a tail, and keep it. */ + a.nset = 2; + memset(sb, 0, sizeof sb); + memset(tb, 0, sizeof tb); + expect("two sets do get a texture issue", + xpsb_attrib_issues(&a, &n, NULL), 1u); + expect("as a tail, so three issues", n, 3u); + xpsb_attrib_bases(&a, &c, sb, tb); + expect("with the texel behind both sets", tb[0], 8u); + } + + expect("volume level 1 follows level 0", + xpsb_twiddle3_level_offset(8, 4, 2, 1), 64u); + expect("volume levels floor every axis at one", + xpsb_twiddle3_level_offset(8, 4, 2, 3), 64u + 8u + 2u); + { + /* A round trip through both orders restores the volume. */ + unsigned char lin[4 * 2 * 2], tw[sizeof lin], back[sizeof lin]; + unsigned i; + + for (i = 0; i < sizeof lin; i++) + lin[i] = (unsigned char)(i * 7u + 3u); + expect("twiddle3 morton", + xpsb_twiddle3_level(tw, lin, 4, 2, 2, 4, 8, 1, + XPSB_TWIDDLE3_MORTON) != 0, 0); + expect("untwiddle3 morton", + xpsb_untwiddle3_level(back, tw, 4, 2, 2, 4, 8, 1, + XPSB_TWIDDLE3_MORTON) != 0, 0); + expect("round trip morton", memcmp(lin, back, sizeof lin) != 0, + 0); + expect("texel (1,0,1) sits at morton index 6", tw[6], lin[9]); + } + + /* The border map. The face offsets are EURASIA_TAG_BORDERMAP_OFFSET_* + * verbatim (sgxdefs.h:5064-5075); the whole map is that plus the + * face's own 8n texels, which is the next entry in the table. */ + { + static const uint32_t ddk[][2] = { + { 1, 56u }, { 2, 64u }, { 4, 80u }, { 8, 112u }, + { 16, 176u }, { 32, 304u }, { 64, 560u }, + { 128, 1072u }, { 256, 2096u }, { 512, 4144u }, + { 1024, 8240u }, { 2048, 16432u } + }; + unsigned q; + + for (q = 0; q < sizeof ddk / sizeof ddk[0]; q++) { + char nm[72]; + + snprintf(nm, sizeof nm, "border face offset %ux%u", + ddk[q][0], ddk[q][0]); + expect(nm, xpsb_border_map_face_offset(ddk[q][0]), + ddk[q][1]); + snprintf(nm, sizeof nm, "border map %ux%u", + ddk[q][0], ddk[q][0]); + expect(nm, xpsb_border_map_texels(ddk[q][0]), + ddk[q][1] + 8u * ddk[q][0]); + /* The table is one prefix sum: a size's whole map ends + * exactly where the next size's face begins. */ + if (q + 1 < sizeof ddk / sizeof ddk[0]) { + snprintf(nm, sizeof nm, "border map %ux%u ends " + "at the next face", ddk[q][0], + ddk[q][0]); + expect(nm, xpsb_border_map_texels(ddk[q][0]), + ddk[q + 1][1]); + } + } + } + expect("border map stops at the texture size cap", + xpsb_border_map_texels(4096), 0u); + expect("border map has no NPOT entry", xpsb_border_map_texels(48), 0u); + + memset(&s, 0, sizeof s); + s.w = 16; s.h = 16; s.format = XPSB_FMT_8888; s.twiddled = 1; + s.umode = XPSB_WRAP_CLAMP_BORDER_MAP; + expect("a map mode without a map is refused", + xpsb_tex_state(&w0, &w1, &s) == 0, 0); + s.border_map = 1; + expect("with one it builds", xpsb_tex_state(&w0, &w1, &s) != 0, 0); + expect("CLAMPBDRMEM is code 6 in U", (w0 >> 6) & 7u, 6u); + s.vmode = XPSB_WRAP_REPEAT_BORDER_MAP; + xpsb_tex_state(&w0, &w1, &s); + expect("REPEATBDRMEM is code 4 in V", (w0 >> 3) & 7u, 4u); + s.h = 8; + expect("a map mode on a non-square texture is refused", + xpsb_tex_state(&w0, &w1, &s) == 0, 0); + s.h = 16; s.twiddled = 0; s.stride = 4 * 32; + expect("or a linear one", xpsb_tex_state(&w0, &w1, &s) == 0, 0); + s.umode = s.vmode = XPSB_WRAP_CLAMP_BORDER; + s.border_map = 0; + expect("CLAMPBDR needs no map", xpsb_tex_state(&w0, &w1, &s) != 0, 0); + expect("and is code 5 in both", (w0 >> 3) & 0x3fu, 5u | (5u << 3)); + + memset(&s, 0, sizeof s); + s.w = 64; s.h = 16; s.format = XPSB_FMT_8888; s.stride = 4 * 64; + s.gamma = 1; + xpsb_tex_state(&w0, &w1, &s); + expect("gamma is bit 27 alone", w0, 0x03fe0000u | (1u << 27)); + s.gamma = 0; s.format = XPSB_FMT_AL88; s.stride = 2 * 64; + xpsb_tex_state(&w0, &w1, &s); + expect("AL88 has no bit 30 by default", w0 & 0x40000000u, 0u); + s.chanrep = 1; + xpsb_tex_state(&w0, &w1, &s); + expect("chanrep sets it", w0 & 0x40000000u, 0x40000000u); +} + +/* Step 6. Blending is two things and only two: enable bit 25 of the per-draw + * ISP word, and the fragment slot. Every operator is held against + * xpsb_composite_shader(), and the heap is held against the unblended one. */ +/* The cull group is not in the capture, so switching it on is a layout change: + * the mask gains bit 12, the coordinate-set word moves up a dword and the + * descriptor grows. Off has to put all three back exactly, or every frame that + * never culls would drift away from the capture. */ +static void test_cull(void) +{ + uint32_t *a = calloc(HEAP_DW, 4), *b = calloc(HEAP_DW, 4); + unsigned i, diff = 0; + + puts("the cull group the capture does not carry:"); + xpsb_gen_heap(a, 128, 128, 0, 0); + xpsb_gen_heap(b, 128, 128, 0, 0); + expect("captured mask", a[XPSB_HEAP_STATE], XPSB_STATE_MASK_CAPTURED); + expect("no cull group in the capture", + a[XPSB_HEAP_STATE] & XPSB_STATE_CULL_BIT, 0); + expect("no cull group offset either", + (uint32_t)xpsb_heap_state_off(a, XPSB_STATE_CULL), (uint32_t)-1); + expect("coordinate-set word", a[STATE(a, XPSB_STATE_TEXSIZE)], + 0x00000005u); + expect("captured block length", + a[XPSB_HEAP_STATE_DESC + 1], XPSB_DMA_CTL(XPSB_STATE_DWORDS)); + + xpsb_heap_set_cull(b, 0x00060001u, 1); + expect("mask gains bit 12", b[XPSB_HEAP_STATE], 0x00005f41u); + expect("cull word in place", b[STATE(b, XPSB_STATE_CULL)], 0x00060001u); + expect("cull word at 0x1155", STATE(b, XPSB_STATE_CULL), 0x1155u); + expect("coordinate-set word moved up one", + b[STATE(b, XPSB_STATE_TEXSIZE)], 0x00000005u); + expect("to 0x1156", STATE(b, XPSB_STATE_TEXSIZE), 0x1156u); + expect("block grew by one dword", b[XPSB_HEAP_STATE_DESC + 1], + XPSB_DMA_CTL(XPSB_STATE_DWORDS + 1)); + expect("and stayed home", xpsb_heap_state_base(b), XPSB_HEAP_STATE); + for (i = 0; i < HEAP_DW; i++) + if (a[i] != b[i] && i != XPSB_HEAP_STATE && + i != 0x1155 && i != 0x1156 && i != XPSB_HEAP_STATE_DESC + 1) + diff++; + expect("nothing else in the heap moved", diff, 0); + + /* And back: a caller that turns culling off mid-frame must not leave + * the block a dword long with a stale word in it. */ + xpsb_heap_set_cull(b, 0, 0); + expect_same("off restores the captured block", a, b, HEAP_DW); + + printf(" mask %#06x -> %#06x, %u -> %u dwords\n", 0x4f41u, 0x5f41u, + XPSB_STATE_DWORDS, XPSB_STATE_DWORDS + 1); + free(a); + free(b); +} + +/* The two things that carry a block to the MTE and have to follow its + * size: the DOUTD control word, in bursts of at most sixteen dwords, and + * the USE program that copies and emits exactly that many. Both were fixed + * at the capture's fifteen, which is what handed every longer block over + * short on the part (work/feat-stateblock/fixes/NOTES.md). */ +static void test_state_carry(void) +{ + static const struct { unsigned n; uint32_t ctl; unsigned carried; } + dma[] = { + { 1, 0x80000000u, 1 }, { 15, 0x81c0000eu, 15 }, + { 16, 0x81e0000fu, 16 }, { 17, 0x81000018u, 18 }, + { 18, 0x81000018u, 18 }, { 19, 0x81200019u, 20 }, + { 20, 0x81200019u, 20 }, { 23, 0x8160001bu, 24 }, + { 24, 0x8160001bu, 24 }, + }; + uint32_t *u = calloc(XPSB_USSE_DW, 4); + uint32_t prog[8]; + unsigned i, carried; + + puts("carrying the block to the MTE:"); + for (i = 0; i < sizeof dma / sizeof dma[0]; i++) { + char what[48]; + + snprintf(what, sizeof what, "DMA control for %u", dma[i].n); + expect(what, xpsb_state_dma_ctl(dma[i].n, &carried), dma[i].ctl); + snprintf(what, sizeof what, " carries %u", dma[i].carried); + expect(what, carried, dma[i].carried); + } + expect("the captured fifteen is the captured word", + xpsb_state_dma_ctl(15, NULL), XPSB_DMA_CTL(15)); + expect("a burst never exceeds sixteen", + (xpsb_state_dma_ctl(24, NULL) & 0xfu) + 1u <= 16u, 1); + expect("nothing past the block's maximum", + xpsb_state_dma_ctl(XPSB_STATE_MAX_DWORDS + 2u, NULL), 0); + for (i = 0; i < sizeof dma / sizeof dma[0]; i++) { + char what[48]; + + snprintf(what, sizeof what, "%u carries %u dwords", dma[i].n, + dma[i].carried); + expect(what, xpsb_dma_ctl_dwords(dma[i].ctl), dma[i].carried); + } + /* Home holds what the DMA carries, not what the block is: sixteen is + * the last count home, seventeen the first at the alternate. */ + { + uint32_t *h = calloc(HEAP_DW, 4); + uint32_t w[3] = { 0x01d00105u, 0, 0x0400ffffu }; + unsigned n; + + xpsb_gen_heap(h, 128, 128, 0, 0); + expect("fifteen is home, four blocks", + xpsb_heap_state_base(h) == XPSB_HEAP_STATE && + xpsb_heap_state_regs(h) == 4, 1); + xpsb_heap_set_isp(h, w, NULL); + n = xpsb_state_dwords(h[xpsb_heap_state_base(h)]); + expect("sixteen is home", n == 16 && + xpsb_heap_state_base(h) == XPSB_HEAP_STATE, 1); + xpsb_heap_set_cull(h, 0x00010001u, 1); + n = xpsb_state_dwords(h[xpsb_heap_state_base(h)]); + expect("seventeen is the alternate", n == 17 && + xpsb_heap_state_base(h) == XPSB_HEAP_STATE_ALT, 1); + expect(" carried as eighteen, five blocks", + xpsb_dma_ctl_dwords(h[XPSB_HEAP_STATE_DESC + 1]) == 18 && + xpsb_heap_state_regs(h) == 5, 1); + xpsb_heap_set_cull(h, 0, 0); + expect("sixteen is home again", + xpsb_heap_state_base(h), XPSB_HEAP_STATE); + free(h); + } + + xpsb_gen_usse(u, XPSB_CLEAR_DEFAULT); + expect("the fifteen-dword program is the captured slot 0x1a0", + memcmp(u + xpsb_usse_state_copy_off(15) / 4, u + 0x1a0 / 4, 16), + 0); + expect("two is the captured slot 0x80", + memcmp(u + xpsb_usse_state_copy_off(2) / 4, u + 0x80 / 4, 16), 0); + expect("six is the captured slot 0x180", + memcmp(u + xpsb_usse_state_copy_off(6) / 4, u + 0x180 / 4, 16), + 0); + expect("eleven is the captured slot 0xe0", + memcmp(u + xpsb_usse_state_copy_off(11) / 4, u + 0xe0 / 4, 16), + 0); + expect("sixteen is one instruction of sixteen", + xpsb_usse_state_copy(prog, 16), 16); + expect(" mov.repeat16 o0, pa0", prog[1], 0x28a1f001u); + expect(" emitst #16", prog[2], 0xa0200800u); + expect("twenty is two moves and the emit", + xpsb_usse_state_copy(prog, 20), 24); + expect(" mov.repeat16 o0, pa0", prog[1], 0x28a1f001u); + expect(" mov.repeat4 o16, pa16", prog[3], 0x28a13001u); + expect(" from o16 and pa16", prog[2], 0xa2000800u); + expect(" emitst.end.freep #0, #20", prog[4], 0xa0200a00u); + expect(" its high word", prog[5], 0xfb274000u); + expect("sizes", xpsb_usse_state_copy_size(16) == 16 && + xpsb_usse_state_copy_size(17) == 24, 1); + expect("twenty-five is refused", xpsb_usse_state_copy(prog, 25), 0); + expect("the programs end below the fragment slots' region", + XPSB_USSE_STATE_END <= XPSB_USSE_DW * 4, 1); + expect("and above the captured slots", XPSB_USSE_STATE_OFF >= 0x2060, 1); + + /* the DOUTU's three relocations follow the config */ + { + struct xpsb_reloc_cfg c; + struct xpsb_reloc r[XPSB_MAX_TA_RELOCS]; + unsigned n; + + memset(&c, 0, sizeof c); + c.ntex = 1; + n = xpsb_gen_ta_relocs(r, &c); + expect("captured DOUTU program", n > 72 && r[70].pre_add == 0x1a0 && + r[71].pre_add == 0x1a0 && r[72].pre_add == 0x1a0, 1); + expect(" and size", r[70].arg0, 0x10); + c.state_use_off = xpsb_usse_state_copy_off(20); + c.state_use_size = xpsb_usse_state_copy_size(20); + xpsb_gen_ta_relocs(r, &c); + expect("the twenty-dword program", r[70].pre_add == c.state_use_off && + r[71].pre_add == c.state_use_off && + r[72].pre_add == c.state_use_off, 1); + expect(" with its size", r[72].arg0, 24); + expect(" at the DOUTU", r[70].where == XPSB_HEAP_STATE_USE && + r[72].where == XPSB_HEAP_STATE_USE, 1); + } + free(u); +} + +/* The block's layout as a whole: a group's dword follows from the mask, an + * ISP word B or C goes in right after A and moves everything above it, and a + * block that outgrows the seventeen dwords before its descriptor moves past + * it with the descriptor's pointer following. Every path back has to restore + * the capture exactly. */ +static void test_state_layout(void) +{ + uint32_t *a = calloc(HEAP_DW, 4), *b = calloc(HEAP_DW, 4); + uint32_t ff[3], bf[3], words[6]; + unsigned i, g, n, diff = 0; + + puts("the state block's layout:"); + xpsb_gen_heap(a, 128, 128, 0, 0); + xpsb_gen_heap(b, 128, 128, 0, 0); + n = 0; + for (g = 0; g < XPSB_STATE_NGROUPS; g++) + n += xpsb_state_size(g); + expect("group sizes sum to the DDK's 23", n, 23); + expect("the captured mask is fifteen dwords", + xpsb_state_dwords(XPSB_STATE_MASK_CAPTURED), XPSB_STATE_DWORDS); + expect("every group present is twenty-four", xpsb_state_dwords(0xffffu), + XPSB_STATE_MAX_DWORDS); + expect("home", xpsb_heap_state_base(a), XPSB_HEAP_STATE); + expect("ISP word A", ISP_A(a), 0x1148u); + expect("PDS pointers", PDS_SEC(a), 0x1149u); + expect("viewport", STATE(a, XPSB_STATE_VIEWPORT), 0x114cu); + expect("wrap", STATE(a, XPSB_STATE_WRAP), 0x1152u); + expect("output selects", STATE(a, XPSB_STATE_OUTSEL), 0x1153u); + expect("w clamp", STATE(a, XPSB_STATE_WCLAMP), 0x1154u); + expect("coordinate sets", STATE(a, XPSB_STATE_TEXSIZE), 0x1155u); + expect("a group not present has no offset", + (uint32_t)xpsb_heap_state_off(a, XPSB_STATE_ISP_C), (uint32_t)-1); + expect("nor one past the mask", + (uint32_t)xpsb_heap_state_off(a, 16), (uint32_t)-1); + + /* Word C alone: the stencil test. */ + ff[0] = a[ISP_A(a)] | XPSB_ISP_CPRES | 0x05u; + ff[1] = 0; + ff[2] = 0x0e00ff00u; + expect("A + C installs", xpsb_heap_set_isp(b, ff, NULL), 0); + expect("mask gains bit 2", b[XPSB_HEAP_STATE], 0x00004f45u); + expect("A carries CPRES and the reference", b[ISP_A(b)], ff[0]); + expect("C follows A", STATE(b, XPSB_STATE_ISP_C), 0x1149u); + expect("C is the word", b[STATE(b, XPSB_STATE_ISP_C)], 0x0e00ff00u); + expect("the PDS pointers moved up one", PDS_SEC(b), 0x114au); + expect("and kept their words", b[PDS_SEC(b)], a[PDS_SEC(a)]); + expect("as did the viewport", b[STATE(b, XPSB_STATE_VIEWPORT) + 3], + a[STATE(a, XPSB_STATE_VIEWPORT) + 3]); + expect("sixteen dwords", b[XPSB_HEAP_STATE_DESC + 1], + XPSB_DMA_CTL(XPSB_STATE_DWORDS + 1)); + expect("still home", xpsb_heap_state_base(b), XPSB_HEAP_STATE); + expect("the descriptor still points home", + b[XPSB_HEAP_STATE_DESC], a[XPSB_HEAP_STATE_DESC]); + for (i = 0; i < HEAP_DW; i++) + if (a[i] != b[i] && (i < XPSB_HEAP_STATE || i > 0x1159)) + diff++; + expect("nothing outside the block moved", diff, 0); + + /* B and C: seventeen dwords. They would fit before the descriptor, + * but the DMA carries eighteen and the part takes that only from + * the alternate - so seventeen is the first count to move. */ + ff[0] |= XPSB_ISP_BPRES; + ff[1] = 0x000800f0u; + expect("A + B + C installs", xpsb_heap_set_isp(b, ff, NULL), 0); + expect("mask", b[XPSB_HEAP_STATE_ALT], 0x00004f47u); + expect("seventeen dwords, carried as eighteen", + b[XPSB_HEAP_STATE_DESC + 1], xpsb_dma_ctl(18, 0)); + expect("moved to the alternate", xpsb_heap_state_base(b), + XPSB_HEAP_STATE_ALT); + expect("home is zeroed", b[XPSB_HEAP_STATE] | b[XPSB_HEAP_STATE + 16], 0); + expect("the descriptor follows", b[XPSB_HEAP_STATE_DESC], + (a[XPSB_HEAP_STATE_DESC] & ~0xffu) | 0x9cu); + expect("five register blocks", xpsb_heap_state_regs(b), 5); + expect("B after A", b[STATE(b, XPSB_STATE_ISP_B)], 0x000800f0u); + expect("C after B", b[STATE(b, XPSB_STATE_ISP_C)], 0x0e00ff00u); + expect("coordinate sets kept", b[STATE(b, XPSB_STATE_TEXSIZE)], 5); + expect("the DMA's extra dword stays inside the window", + XPSB_HEAP_STATE_ALT + 18 <= XPSB_HEAP_STATE_END, 1); + + /* The cull group on top: eighteen exactly. */ + xpsb_heap_set_cull(b, 0x00010001u, 1); + expect("mask", b[XPSB_HEAP_STATE_ALT], 0x00005f47u); + expect("eighteen dwords, two bursts of nine", + b[XPSB_HEAP_STATE_DESC + 1], 0x81000018u); + expect("still the alternate", xpsb_heap_state_base(b), + XPSB_HEAP_STATE_ALT); + expect("the cull word", b[STATE(b, XPSB_STATE_CULL)], 0x00010001u); + expect("C still after B", b[STATE(b, XPSB_STATE_ISP_C)], 0x0e00ff00u); + + /* A back-face set on top: twenty-one. */ + ff[0] |= XPSB_ISP_2SIDED; + bf[0] = ff[0] | 0x03u; + bf[1] = ff[1]; + bf[2] = 0x0400ff00u; + expect("two-sided installs", xpsb_heap_set_isp(b, ff, bf), 0); + expect("mask", b[XPSB_HEAP_STATE_ALT], 0x00005f7fu); + expect("twenty-one dwords, three bursts of seven", + b[XPSB_HEAP_STATE_DESC + 1], 0x80c00026u); + expect("back A", b[STATE(b, XPSB_STATE_ISP_BF_A)], bf[0]); + expect("back B", b[STATE(b, XPSB_STATE_ISP_BF_B)], bf[1]); + expect("back C", b[STATE(b, XPSB_STATE_ISP_BF_C)], bf[2]); + expect("front C before it", STATE(b, XPSB_STATE_ISP_C) + 1, + STATE(b, XPSB_STATE_ISP_BF_A)); + expect("PDS pointers after", PDS_SEC(b), XPSB_HEAP_STATE_ALT + 7); + expect("with their words", b[PDS_PRI(b)], a[PDS_PRI(a)]); + expect("viewport words", b[STATE(b, XPSB_STATE_VIEWPORT) + 5], + a[STATE(a, XPSB_STATE_VIEWPORT) + 5]); + expect("coordinate sets last", b[STATE(b, XPSB_STATE_TEXSIZE)], 5); + expect("inside the window", STATE(b, XPSB_STATE_TEXSIZE) < + XPSB_HEAP_STATE_END, 1); + /* the record width still lands in the moved block */ + expect("record width sets", xpsb_heap_set_record_width(b, 14, 0), 0); + expect("in the moved group 10", + (b[STATE(b, XPSB_STATE_OUTSEL)] & XPSB_STATE_G10_DW) >> + XPSB_STATE_G10_SHIFT, 14); + xpsb_heap_set_record_width(b, 11, 0); + + /* every group at once fits the alternate */ + memset(words, 0x11, sizeof words); + expect("region clip fits", xpsb_heap_state_set(b, XPSB_STATE_RGNCLIP, + 1, words), 0); + expect("texfloat fits", xpsb_heap_state_set(b, XPSB_STATE_TEXFLOAT, 1, + words), 0); + expect("terminate fits", xpsb_heap_state_set(b, XPSB_STATE_TERMINATE, + 1, words), 0); + expect("twenty-four dwords, two bursts of twelve", + b[XPSB_HEAP_STATE_DESC + 1], 0x8160001bu); + expect("a group the block does not know is refused", + xpsb_heap_state_set(b, 16, 1, words) != 0, 1); + expect("as is a group with no words", + xpsb_heap_state_set(b, XPSB_STATE_ISP_B, 1, NULL) != 0, 1); + xpsb_heap_state_set(b, XPSB_STATE_RGNCLIP, 0, NULL); + xpsb_heap_state_set(b, XPSB_STATE_TEXFLOAT, 0, NULL); + xpsb_heap_state_set(b, XPSB_STATE_TERMINATE, 0, NULL); + + /* And back, in the other order, through home. */ + ff[0] &= ~XPSB_ISP_2SIDED; + expect("one-sided again", xpsb_heap_set_isp(b, ff, NULL), 0); + expect("eighteen, still the alternate", xpsb_heap_state_base(b), + XPSB_HEAP_STATE_ALT); + xpsb_heap_set_cull(b, 0, 0); + expect("seventeen, still the alternate", xpsb_heap_state_base(b), + XPSB_HEAP_STATE_ALT); + ff[0] &= ~XPSB_ISP_BPRES; + expect("no B", xpsb_heap_set_isp(b, ff, NULL), 0); + expect("sixteen, back home", xpsb_heap_state_base(b), XPSB_HEAP_STATE); + expect("alternate zeroed", + b[XPSB_HEAP_STATE_ALT] | b[XPSB_HEAP_STATE_ALT + 20], 0); + expect("the descriptor points home", b[XPSB_HEAP_STATE_DESC], + a[XPSB_HEAP_STATE_DESC]); + expect("four register blocks, as captured", xpsb_heap_state_regs(b), 4); + ff[0] = a[ISP_A(a)]; + expect("plain A again", xpsb_heap_set_isp(b, ff, NULL), 0); + expect_same("the capture is back", a, b, HEAP_DW); + + /* The ISP words on a record's copy of the window, which is what the + * driver does: the same code on a pointer offset by whole windows. */ + memcpy(b + 0x2000, a + 0x1140, 0x40 * 4); + ff[0] = a[ISP_A(a)] | XPSB_ISP_CPRES | XPSB_ISP_2SIDED; + ff[2] = 0x0e00ff00u; + bf[0] = ff[0]; + bf[1] = 0; + bf[2] = 0x0400ff00u; + expect("two-sided on a copy", + xpsb_heap_set_isp(b + 0x2000 - 0x1140, ff, bf), 0); + expect("laid out past the copy's descriptor", + xpsb_heap_state_base(b + 0x2000 - 0x1140), XPSB_HEAP_STATE_ALT); + expect("the copy's descriptor low byte", b[0x2000 + 0x18] & 0xffu, + 0x9cu); + expect("the copy's mask", b[0x2000 + 0x27], 0x00004f7du); + expect("the original untouched", xpsb_heap_state_base(a), + XPSB_HEAP_STATE); + free(a); + free(b); +} + +static void test_blend(void) +{ + static const struct xpsb_shader_flags fl[4] = { + { 0, 0, 0, 0 }, { 1, 0, 0, 0 }, { 0, 1, 0, 1 }, { 0, 0, 1, 0 } + }; + uint32_t *a = calloc(HEAP_DW, 4), *b = calloc(HEAP_DW, 4); + uint32_t slot[XPSB_USSE_SLOT_DW], want[4]; + unsigned i, k, diff = 0; + int op; + + puts("blending and the composite shader:"); + xpsb_gen_heap(a, 128, 128, 0, 0); + xpsb_gen_heap(b, 128, 128, 0, 0); + expect("captured ISP word", a[ISP_A(a)], 0x01d00000u); + expect("blend bit clear by default", a[ISP_A(a)] & XPSB_ISP_BLEND, 0); + + xpsb_heap_set_blend(b, 1); + expect("blend bit set", b[ISP_A(b)] & XPSB_ISP_BLEND, + XPSB_ISP_BLEND); + expect("ISP word is the documented blend-on value", b[ISP_A(b)], + 0x03d00000u); + for (i = 0; i < HEAP_DW; i++) + if (a[i] != b[i] && i != ISP_A(a)) { + printf(" FAIL blend moved %#06x\n", i); + diff++; + } + expect("only the ISP word moves", diff, 0); + xpsb_heap_set_blend(b, 0); + expect_same("blend off restores the heap", a, b, HEAP_DW); + + for (op = 0; op < XPSB_NUM_OPS; op++) + for (k = 0; k < 4; k++) { + char what[56]; + + expect("usse_composite ok", + xpsb_usse_composite(slot, op, &fl[k]) != 0, 0); + xpsb_composite_shader(op, &fl[k], want); + snprintf(what, sizeof what, "op %d flags %u words", op, k); + expect_same(what, slot, want, 4); + snprintf(what, sizeof what, "op %d flags %u tail", op, k); + expect(what, slot[4] | slot[5] | slot[6] | slot[7], 0); + } + expect("operator 14 refused", xpsb_usse_composite(slot, 14, &fl[0]) == 0, 0); + expect("operator -1 refused", xpsb_usse_composite(slot, -1, &fl[0]) == 0, 0); + expect("no flags refused", xpsb_usse_composite(slot, 0, NULL) == 0, 0); + + xpsb_usse_default_frag(slot); + expect("default fragment low", slot[0], XPSB_FRAG_DEFAULT_LO); + expect("default fragment high", slot[1], XPSB_FRAG_DEFAULT_HI); + printf(" %d operators x 4 flag sets, ISP %#010x -> %#010x\n", + XPSB_NUM_OPS, 0x01d00000u, 0x03d00000u); + free(a); + free(b); +} + +/* The USSE buffer is generated, not replayed, so the fragment slot the + * composite path overwrites has to be the one drmcube puts there. */ +static void test_usse(void) +{ + uint32_t *u = calloc(XPSB_USSE_DW, 4); + uint32_t slot[XPSB_USSE_SLOT_DW]; + unsigned i, n, nonzero = 0; + + puts("USSE programs:"); + n = xpsb_gen_usse(u, XPSB_CLEAR_DEFAULT); + expect("dword count", n, XPSB_USSE_DW); + expect("pixel emit low", u[0], 0x84208180u); + expect("pixel emit high", u[1], 0xfb200004u); + expect("clear colour immediate", u[0x40 / 4], XPSB_CLEAR_DEFAULT); + expect("clear colour high", u[0x44 / 4], 0xfca7f181u); + expect("clear immediate clamped to 21 bits", + (xpsb_gen_usse(u, 0xffffffffu), u[0x40 / 4]), 0x1fffffu); + n = xpsb_gen_usse(u, XPSB_CLEAR_DEFAULT); + expect("fragment low", u[XPSB_USSE_FRAG_OFF / 4], XPSB_FRAG_DEFAULT_LO); + expect("fragment high", u[XPSB_USSE_FRAG_OFF / 4 + 1], + XPSB_FRAG_DEFAULT_HI); + expect("second pixel emit at 0x2000", u[0x2000 / 4], 0x84208180u); + expect("last program at 0x2040", u[0x2044 / 4], 0x28851001u); + for (i = 0; i < XPSB_USSE_DW; i++) + if (u[i]) nonzero++; + expect("nonzero dwords", nonzero, 48 + 16 * 4 + 8 * 6); + + /* the composite shader may reach the fragment slot and nothing else */ + { + struct xpsb_shader_flags fl; + uint32_t *v = calloc(XPSB_USSE_DW, 4); + unsigned diff = 0; + + memset(&fl, 0, sizeof fl); + xpsb_usse_composite(slot, XPSB_OP_OVER, &fl); + memcpy(u + XPSB_USSE_FRAG_OFF / 4, slot, sizeof slot); + + xpsb_gen_usse(v, XPSB_CLEAR_DEFAULT); + for (i = 0; i < XPSB_USSE_DW; i++) + if (u[i] != v[i] && + (i < XPSB_USSE_FRAG_OFF / 4 || + i >= XPSB_USSE_FRAG_OFF / 4 + XPSB_USSE_SLOT_DW)) + diff++; + expect("only the fragment slot moves", diff, 0); + free(v); + } + printf(" %u slots, %u dwords, fragment at %#06x\n", + XPSB_USSE_NSLOT, XPSB_USSE_DW, XPSB_USSE_FRAG_OFF); + free(u); +} + +/* Step 7. The table at two units must be the one at one unit plus exactly the + * unit-1 address record and the primary binding's new data size; the heap must + * move only the dwords a second sampled unit names, and the PDS program those + * dwords carry must be the one xpsb_pds.c generates. */ +static void test_two_textures(void) +{ + /* corpus P_mt3, whose coordinate-set nibbles are 1 and 2 where this + * stream's vertex record carries sets 0 and 1 */ + static const uint32_t mt3_data[16] = { + 0x0018102c, 0x00000000, 0x03fe0090, 0x6c01f01f, + 0x03fe0090, 0x6c00f00f, 0x00000000, 0x00000000, + 0x00000020, 0x0fc0a201, 0x804ee000, 0x0c00fa02, + 0x804ef000, 0x00000000, 0x00000000, 0x00000000, + }; + static const uint32_t mt3_code[6] = { + 0x07000345, 0x070418a2, 0x07042364, 0x070438a2, + 0x07084364, 0xaf000000, + }; + /* every heap dword xpsb_heap_set_ntex() is allowed to move */ + static const unsigned moved[] = { + 0x0d9, 0x0db, 0x0dc, 0x0dd, 0x0de, 0x0df, 0x0e0, 0x0e1, + 0x0e2, 0x0e3, 0x0e4, 0x0e5, 0x0e9, 0x0f0, 0x114a, 0x114b, + }; + struct xpsb_reloc one[XPSB_NUM_TA_RELOCS], two[XPSB_NUM_TA_RELOCS + 1]; + struct xpsb_pds_tex tex[XPSB_NTEX_MAX]; + uint32_t data[XPSB_PDS_MAX_DATA], code[XPSB_PDS_MAX_CODE], use[3]; + uint32_t *a = calloc(HEAP_DW, 4), *b = calloc(HEAP_DW, 4); + uint32_t *ua = calloc(XPSB_USSE_DW, 4); + uint32_t *da = calloc(MISC_DW, 4), *db = calloc(MISC_DW, 4); + float v[4 * XPSB_VTX_STRIDE_MAX]; + struct xpsb_quad q = { 10, 20, 74, 84, 0, 0, 1, 1, 1, 1, 1, 1, + 0.25f, 0.5f, 0.75f, 1.0f }; + struct xpsb_reloc_cfg c; + unsigned i, j, n1, n2, nd, nc, off, diff = 0, k; + + puts("second texture unit:"); + memset(&c, 0, sizeof c); + c.ntex = 1; + n1 = xpsb_gen_ta_relocs(one, &c); + c.ntex = 2; + c.tex_offset[1] = 0x1000; + n2 = xpsb_gen_ta_relocs(two, &c); + expect("one unit is still 77", n1, XPSB_NUM_TA_RELOCS); + expect("one unit still byte-identical", + memcmp(one, ta_relocs, sizeof one) != 0, 0); + expect("two units is 78", n2, XPSB_NUM_TA_RELOCS + 1); + expect("three units refused", + (c.ntex = 3, xpsb_gen_ta_relocs(two, &c)), 0); + c.ntex = 2; + xpsb_gen_ta_relocs(two, &c); + + /* record 48 is unit 0's; 49 is the addition, and every record after it + * is the one-unit table shifted by one, bar the primary binding */ + for (i = 0; i < 49; i++) + if (memcmp(&two[i], &one[i], sizeof one[0])) { + printf(" FAIL record %u moved before the addition\n", i); + diff++; + } + for (i = 50; i < n2; i++) + if (i != 69 && memcmp(&two[i], &one[i - 1], sizeof one[0])) { + printf(" FAIL record %u moved after the addition\n", i); + diff++; + } + expect("one record added, one changed", diff, 0); + expect("unit 0 where", two[48].where, XPSB_TEXADDR_DW); + expect("unit 0 buffer", two[48].buffer, XPSB_BUF_TEX); + expect("unit 1 where", two[49].where, XPSB_TEXADDR_DW + 2); + expect("unit 1 where is byte 0x370", (XPSB_TEXADDR_DW + 2) * 4, 0x370); + expect("unit 1 buffer", two[49].buffer, 8); + expect("unit 1 offset", two[49].pre_add, 0x1000); + expect("unit 1 op", two[49].reloc_op, XPSB_RELOC_OP_OFFSET); + expect("unit 1 mask", two[49].mask, 0xffffffffu); + + /* the primary binding declares the data-segment size, so it is the one + * record whose contents follow the unit count */ + expect("primary binding where", one[68].where, 0x114bu); + expect("primary binding at one unit", one[68].background, 12u << 24); + expect("primary binding where at two", two[69].where, 0x114bu); + expect("primary binding at two units", two[69].background, 16u << 24); + two[69].background = one[68].background; + expect("nothing else in that record moves", + memcmp(&two[69], &one[68], sizeof one[0]) != 0, 0); + + /* the secondary binding is the record that has to leave heap+0x380 */ + expect("secondary binding where", two[68].where, 0x1149u); + expect("secondary still at the bare HALT", two[68].pre_add, + XPSB_SEC_PDS_NULL); + c.sec_pds_off = XPSB_SEC_PDS_OFF; + xpsb_gen_ta_relocs(two, &c); + expect("secondary follows the displacement", two[68].pre_add, + XPSB_SEC_PDS_OFF); + expect("displaced secondary declares no data", two[68].background, 0); + c.sec_pds_off = 0; + xpsb_gen_ta_relocs(two, &c); + + expect("descriptor step", XPSB_TEX_UNIT_STEP, 8); + expect("unit 1 control at 0x350", XPSB_TEXCTL_OFF + XPSB_TEX_UNIT_STEP, + 0x350); + expect("unit 1 state at 0x354", XPSB_TEXSTATE_OFF + XPSB_TEX_UNIT_STEP, + 0x354); + + expect("vertex floats at one unit", xpsb_vtx_stride(1), 11); + expect("vertex floats at two units", xpsb_vtx_stride(2), 14); + expect("vertex bytes 44 -> 56", xpsb_vtx_stride(2) * 4, 56); + xpsb_gen_heap(a, 128, 128, 0, 1); + xpsb_gen_heap(b, 128, 128, 0, 1); + expect("captured stride is 44", a[XPSB_VTXSTRIDE_DW], 44); + expect("captured DMA control", a[XPSB_VTXDMA_DW], XPSB_DMA_CTL(11)); + expect("captured varying count", a[PDS_CTL(a)] & 0xf, 2); + expect("captured primary data size", a[PDS_PRI(a)] >> 24, 12); + expect("captured bare secondary program", + a[XPSB_SEC_PDS_NULL / 4], XPSB_PDS_HALT); + + xpsb_heap_set_ntex(b, 2); + expect("stride becomes 56", b[XPSB_VTXSTRIDE_DW], 56); + expect("DMA control becomes 14 dwords", b[XPSB_VTXDMA_DW], + XPSB_DMA_CTL(14)); + expect("DMA control carries the count twice", + ((b[XPSB_VTXDMA_DW] >> 21) & 0x3ff) | ((b[XPSB_VTXDMA_DW] & 0xff) << 16), + 13u | (13u << 16)); + expect("varying count becomes 3", b[PDS_CTL(b)] & 0xf, 3); + expect("primary data size becomes 16", b[PDS_PRI(b)] >> 24, 16); + expect("secondary attribute blocks untouched", + b[PDS_CTL(b)] & XPSB_SA_BLOCK_MASK, + a[PDS_CTL(a)] & XPSB_SA_BLOCK_MASK); + /* the code segment now covers the bare secondary program, which is why + * the frame layer moves it to XPSB_SEC_PDS_OFF */ + expect("the code segment reaches heap+0x380", + b[XPSB_SEC_PDS_NULL / 4], 0x07000345u); + + for (i = 0; i < HEAP_DW; i++) { + int allowed = 0; + + if (a[i] == b[i]) continue; + for (j = 0; j < sizeof moved / sizeof moved[0]; j++) + if (moved[j] == i) allowed = 1; + if (!allowed) { + printf(" FAIL heap dword %#06x moved: %#010x -> %#010x\n", + i, a[i], b[i]); + diff++; + } + } + expect("only the named heap dwords move", diff, 0); + xpsb_heap_set_ntex(b, 1); + expect_same("one unit restores the heap", a, b, HEAP_DW); + + /* the dwords those 22 heap words carry are xpsb_pds.c's program, and + * that program is corpus P_mt3 bar the coordinate-set nibbles */ + xpsb_heap_set_ntex(b, 2); + use[0] = b[XPSB_PRI_PDS_DW]; + use[1] = b[XPSB_PRI_PDS_DW + 1]; + use[2] = b[XPSB_PRI_PDS_DW + 8]; + for (k = 0; k < 2; k++) { + tex[k].ctl = b[(XPSB_TEXCTL_OFF + k * XPSB_TEX_UNIT_STEP) / 4]; + tex[k].fmt = b[(XPSB_TEXSTATE_OFF + k * XPSB_TEX_UNIT_STEP) / 4]; + tex[k].addr = b[XPSB_PRI_PDS_DW + 10 + 2 * k]; + tex[k].itr = b[XPSB_PRI_PDS_DW + 9 + 2 * k]; + } + /* Three live sets, built into the heap: the program the part is given. + * The fourth issue a tail dummy needs overruns the captured slot's + * data segment, so the program is built in the spare block instead of + * shedding the tail - a pixel task with no TAG issue does not batch + * and, where the program samples for itself, has nothing to wait on. + * This holds what the part actually runs today, so a change to it is + * visible. */ + { + uint32_t *c3 = calloc(HEAP_DW, 4); + struct xpsb_attribs at; + unsigned pri = 0, q, ntex = 0, base_dw; + + xpsb_gen_heap(c3, 128, 128, 0, 1); + memset(&at, 0, sizeof at); + at.nset = 3; + for (q = 0; q < 3; q++) { + at.set[q].width = 4; + at.set[q].iterated = 1; + } + expect("three sets build a primary program", + (uint32_t)xpsb_heap_set_attribs(c3, &at, &pri), 0); + /* Four issues do not fit the captured slot's sixteen data + * dwords, so rather than shed the tail the program is built + * in the spare block, which takes twenty-eight. */ + expect("in the spare block", xpsb_pri_pds_off(&at), + (uint32_t)XPSB_PRI_PDS_ALT); + expect("of twenty-eight data dwords", pri, 28u); + expect("the record is twenty floats", xpsb_attrib_stride(&at), + 20u); + expect("group 14 calls all three UVST", + c3[xpsb_heap_state_off(c3, XPSB_STATE_TEXSIZE)], + 0x1ffu); + /* ds1[1 + 2i] is issue i's iterator word, read where the + * program was actually built. */ + base_dw = xpsb_pri_pds_off(&at) / 4u; + for (q = 0; q < 4; q++) { + uint32_t w = c3[base_dw + + xpsb_pds_ds_dword(1, 1 + 2 * q)]; + + if ((w & 0xfu) != 0xfu) + ntex++; + } + expect("and the tail is the one TAG issue", ntex, 1u); + free(c3); + } + + + expect("generator accepts the heap's own words", + xpsb_pds_gen_primary(2, tex, use, data, &nd, code, &nc, &off), 0); + expect("data segment 16 dwords", nd, 16); + expect("code segment 6 dwords", nc, 6); + expect("code segment at heap+0x380", XPSB_PRI_PDS_OFF + off, 0x380); + expect_same("heap data segment is the generator's", + b + XPSB_PRI_PDS_DW, data, nd); + expect_same("heap code segment is the generator's", + b + XPSB_PRI_PDS_DW + nd, code, nc); + expect_same("code segment is corpus P_mt3", code, mt3_code, nc); + diff = 0; + for (i = 0; i < 16; i++) + if (data[i] != mt3_data[i]) { + /* the USE task word, the texture state and the address + * placeholders differ per stream; the iterators may + * differ only in the coordinate-set nibble */ + if ((i == 9 || i == 11) && + (data[i] & ~0xfu) == (mt3_data[i] & ~0xfu)) + continue; + if (i == 0 || i == 2 || i == 3 || i == 4 || i == 5 || + i == 10 || i == 12) + continue; + printf(" FAIL data %u: %#010x, P_mt3 has %#010x\n", + i, data[i], mt3_data[i]); + diff++; + } + expect("data segment matches P_mt3 where the streams agree", diff, 0); + expect("unit 0 samples coordinate set 0", data[9] & 0xf, 0); + expect("unit 1 samples coordinate set 1", data[11] & 0xf, 1); + + /* the vertex USE program emits the record the DMA fetched */ + xpsb_gen_usse(ua, XPSB_CLEAR_DEFAULT); + expect("captured vertex emit count", + (ua[XPSB_USSE_VTX_OFF / 4 + 1] & XPSB_USSE_VTX_MASK) >> 12, 10); + xpsb_usse_set_vtx_dwords(ua, 14); + expect("vertex emit count follows the record", + ua[XPSB_USSE_VTX_OFF / 4 + 1], 0x28a1d001u); + xpsb_usse_set_vtx_dwords(ua, 11); + expect("one unit restores it", ua[XPSB_USSE_VTX_OFF / 4 + 1], + 0x28a1a001u); + + /* and the user draw's record counts it in four-dword granules */ + expect("clear draw is one granule", XPSB_DRAW_KIND(1), 0x02000103u); + expect("eight dwords is two granules", XPSB_DRAW_KIND(2), 0x04000203u); + expect("captured user draw is three", XPSB_DRAW_KIND(3), 0x06000303u); + xpsb_gen_draw_records_n(da, 0, 1); + xpsb_gen_draw_records_n(db, 0, 2); + expect("one unit is the captured record", + da[XPSB_USER_DRAW * XPSB_DRAW_STRIDE + 6], 0x06000303u); + expect("two units is four granules", + db[XPSB_USER_DRAW * XPSB_DRAW_STRIDE + 6], XPSB_DRAW_KIND(4)); + db[XPSB_USER_DRAW * XPSB_DRAW_STRIDE + 6] = + da[XPSB_USER_DRAW * XPSB_DRAW_STRIDE + 6]; + expect_same("no other draw dword moves", da, db, MISC_DW); + expect("zero units refused", xpsb_gen_draw_records_n(db, 0, 0), 0); + + /* dword 4 is the DDK's index list word 2; [26:25] of it is the vertex + * data master's flat-shade selector, and the corner is named as + * 1 + it so the MTE, the iterator and the master share one value */ + expect("a gouraud record leaves the master at VERTEX0", + da[XPSB_USER_DRAW * XPSB_DRAW_STRIDE + XPSB_DRAW_IDX_WORD], 0); + expect("gouraud writes nothing", xpsb_vdm_idx_word(0, 0), 0); + expect("a first provoking vertex is VERTEX0", xpsb_vdm_idx_word(0, 1), + 0); + expect("the middle corner is VERTEX1", xpsb_vdm_idx_word(0, 2), + 1u << 25); + expect("a last one is VERTEX2", xpsb_vdm_idx_word(0, 3), 2u << 25); + expect("the index size beside it is untouched", + xpsb_vdm_idx_word(1u << 24, 3), (2u << 25) | (1u << 24)); + expect("and a corner already there is replaced, not merged", + xpsb_vdm_idx_word(2u << 25, 1), 0); + + memset(v, 0, sizeof v); + expect("quad at two units", xpsb_gen_quad_n(v, NULL, &q, 2, NULL), 6); + for (k = 0; k < 4; k++) { + const float *o = v + k * 14; + char what[40]; + static const float wm[4] = { 0.25f, 0.75f, 0.75f, 0.25f }; + static const float wn[4] = { 0.5f, 0.5f, 1.0f, 1.0f }; + + snprintf(what, sizeof what, "vertex %u mask u", k); + expect(what, (uint32_t)(o[11] * 100.0f), (uint32_t)(wm[k] * 100.0f)); + snprintf(what, sizeof what, "vertex %u mask v", k); + expect(what, (uint32_t)(o[12] * 100.0f), (uint32_t)(wn[k] * 100.0f)); + snprintf(what, sizeof what, "vertex %u mask q", k); + expect(what, (uint32_t)o[13], 1); + } + expect("zero units refused", xpsb_gen_quad_n(v, NULL, &q, 0, NULL), 0); + expect("three units refused", xpsb_gen_quad_n(v, NULL, &q, 3, NULL), 0); + printf(" 77 -> %u relocations, vertex 44 -> %u bytes, PDS 12 -> %u dwords\n", + n2, xpsb_vtx_stride(2) * 4, nd); + free(a); + free(b); + free(ua); + free(da); + free(db); +} + +/* The refusal is the fallback trigger for the whole composite path, so it is + * checked on every setter and on a width that is not a multiple of 32. */ +static void test_stride_refusal(void) +{ + struct xpsb_surface_desc s; + uint32_t *heap = calloc(HEAP_DW, 4), w0, w1; + + puts("stride refusal, stride == bpp * ALIGN(w,32):"); + memset(&s, 0, sizeof s); + s.w = 100; s.h = 50; s.format = XPSB_FMT_8888; + + s.stride = 4 * 128; + expect("w=100 padded to 128 ok", xpsb_surface_check(&s) != 0, 0); + s.stride = 4 * 100; + expect("w=100 unpadded refused", xpsb_surface_check(&s) == 0, 0); + expect("tex_state refuses", xpsb_tex_state(&w0, &w1, &s) == 0, 0); + expect("heap_set_dest refuses", xpsb_heap_set_dest(heap, &s) == 0, 0); + s.stride = 4 * 160; + expect("over-padded refused", xpsb_surface_check(&s) == 0, 0); + s.stride = 2 * 128; + expect("wrong bpp refused", xpsb_surface_check(&s) == 0, 0); + s.stride = 4 * 128; s.format = 0x1e; + expect("unknown format refused", xpsb_surface_check(&s) == 0, 0); + s.format = XPSB_FMT_8888; s.w = 0; + expect("zero width refused", xpsb_surface_check(&s) == 0, 0); + free(heap); +} + +/* set_dest must reproduce gen_heap exactly for the surface gen_heap describes, + * and move only the render-target fields for any other one. */ +static void test_dest_heap(void) +{ + uint32_t *a = calloc(HEAP_DW, 4), *b = calloc(HEAP_DW, 4); + struct xpsb_surface_desc s; + unsigned i, diff = 0; + + puts("caller-supplied destination:"); + memset(&s, 0, sizeof s); + s.w = 200; s.h = 100; s.format = XPSB_FMT_8888; s.stride = 4 * 224; + + xpsb_gen_heap(a, 200, 100, 0, 1); + xpsb_gen_heap(b, 200, 100, 0, 1); + expect("set_dest ok", xpsb_heap_set_dest(b, &s) != 0, 0); + expect_same("set_dest is a no-op", a, b, HEAP_DW); + + s.w = 256; s.h = 64; s.stride = 4 * 256; + expect("set_dest 256x64", xpsb_heap_set_dest(b, &s) != 0, 0); + expect("0x000 stride", b[0x000], 255u << 15); + expect("0x003 h|stride", b[0x003], (63u << 12) | 255u); + expect("0x04b w|h", b[0x04b], 0x6c000000u | (255u << 12) | 63u); + expect("0x05a tile shadow", b[0x05a], (16u << 16) | 4u); + expect("0x10033 w|h", b[0x10033], 0x6c000000u | (255u << 12) | 63u); + /* the two surface descriptors, the tile shadow, the twelve size floats, + * the present extent and the viewport quad - and nothing else */ + { + static const unsigned moved[] = { + 0x000, 0x003, 0x04b, 0x05a, + 0x073, 0x078, 0x07b, 0x07c, + 0x10003, 0x10033, + 0x1102, 0x1105, 0x1106, 0x10fd, 0x110d, 0x1112, + 0x1115, 0x1116, + 0x114c, 0x114d, 0x114e, 0x114f + }; + unsigned j; + + for (i = 0; i < HEAP_DW; i++) { + if (a[i] == b[i]) continue; + for (j = 0; j < sizeof moved / sizeof moved[0]; j++) + if (moved[j] == i) break; + if (j == sizeof moved / sizeof moved[0]) + printf(" FAIL unexpected move at %#06x\n", i); + diff++; + } + expect("render-target fields only", diff, + sizeof moved / sizeof moved[0]); + } + + s.format = XPSB_FMT_565; s.stride = 2 * 256; + expect("set_dest 565", xpsb_heap_set_dest(b, &s) != 0, 0); + expect("0x04b 565 top byte", b[0x04b] >> 24, 0x65); + expect("0x10033 565 top byte", b[0x10033] >> 24, 0x65); + free(a); + free(b); +} + +static void test_quad(void) +{ + float v[4 * XPSB_VTX_STRIDE]; + uint16_t idx[6]; + struct xpsb_quad q = { 10, 20, 74, 84, 0, 0, 1, 1, 1, 1, 1, 1, 0, 0, 0, 0 }; + static const float wx[4] = { 10, 74, 74, 10 }; + static const float wy[4] = { 20, 20, 84, 84 }; + static const float wu[4] = { 0, 1, 1, 0 }; + static const float wv[4] = { 0, 0, 1, 1 }; + static const uint16_t wi[6] = { 0, 1, 3, 1, 2, 3 }; + unsigned k, nvtx = 0, nidx; + + puts("one quad:"); + memset(v, 0, sizeof v); + nidx = xpsb_gen_quad(v, idx, &q, &nvtx); + expect("index count", nidx, 6); + expect("vertex count", nvtx, 4); + for (k = 0; k < 4; k++) { + const float *o = v + k * XPSB_VTX_STRIDE; + char what[32]; + + snprintf(what, sizeof what, "vertex %u x", k); + expect(what, (uint32_t)o[0], (uint32_t)wx[k]); + snprintf(what, sizeof what, "vertex %u y", k); + expect(what, (uint32_t)o[1], (uint32_t)wy[k]); + snprintf(what, sizeof what, "vertex %u u", k); + expect(what, (uint32_t)o[8], (uint32_t)wu[k]); + snprintf(what, sizeof what, "vertex %u v", k); + expect(what, (uint32_t)o[9], (uint32_t)wv[k]); + snprintf(what, sizeof what, "vertex %u z/w/q", k); + expect(what, (uint32_t)(o[2] * 2.0f + o[3] + o[10]), 3); + snprintf(what, sizeof what, "vertex %u index", k); + expect(what, idx[k], wi[k]); + } + expect("index 4", idx[4], wi[4]); + expect("index 5", idx[5], wi[5]); +} + +/* Dropping the two clearing draws moves the user draw to slot 0 and the + * terminator to slot 1; nothing else about either record may change. */ +static void test_noclear_records(void) +{ + uint32_t all[MISC_DW], none[MISC_DW], ref[MISC_DW]; + uint32_t rg_all[RAST_DW], rg_none[RAST_DW]; + unsigned i, n; + + puts("draw records and rastgeom without the clear:"); + memset(all, 0, sizeof all); + memset(none, 0, sizeof none); + memset(ref, 0, sizeof ref); + n = xpsb_gen_draw_records_from(all, 0); + gen_draw_records(ref); + expect("record count with clear", n, 4); + expect_same("from(0) == drmcube", all, ref, MISC_DW); + + n = xpsb_gen_draw_records_from(none, XPSB_USER_DRAW); + expect("record count without clear", n, 2); + for (i = 0; i < 7; i++) { + char what[40]; + + snprintf(what, sizeof what, "user draw word %u moved", i); + expect(what, none[i], all[2 * XPSB_DRAW_STRIDE + i]); + } + for (i = 0; i < 4; i++) { + char what[40]; + + snprintf(what, sizeof what, "terminator word %u moved", i); + expect(what, none[XPSB_DRAW_STRIDE + i], + all[3 * XPSB_DRAW_STRIDE + i]); + } + expect("command offset with clear", xpsb_draw_cmd_off(0), + XPSB_DRAW_CMD_OFF); + expect("command offset without", xpsb_draw_cmd_off(XPSB_USER_DRAW), 8); + expect("command word found", none[8 / 4] & 0xfff00000u, + XPSB_DRAW_CMD_TAG); + expect("refuse dropping the user draw", + xpsb_gen_draw_records_from(none, XPSB_USER_DRAW + 1), 0); + + memset(rg_all, 0, sizeof rg_all); + memset(rg_none, 0, sizeof rg_none); + xpsb_gen_rastgeom_from(rg_all, 128, 128, 0); + xpsb_gen_rastgeom_from(rg_none, 128, 128, XPSB_USER_DRAW); + for (i = 0; i < 0x400; i++) + if (rg_none[i]) { + expect("clearing rastgeom cleared", i, 0); + break; + } + expect_same("present rastgeom kept", rg_all + 0x400, + rg_none + 0x400, RAST_DW - 0x400); +} + +/* The regenerated tables must be byte-identical to test/reloc_tables.h before + * any other configuration is believed. */ +/* The constant operand. Every word is restated here rather than taken from a + * header: the two PDS instructions, the binding's two fields, and the heap + * offset the program moves to. */ +static void test_scalar(void) +{ + static const uint32_t douta = 0x07030226, halt = 0xaf000000; + static const uint32_t doutd = 0x07030223; + uint32_t prog[XPSB_SEC_PDS_MAX]; + uint32_t *a = calloc(HEAP_DW, 4), *b = calloc(HEAP_DW, 4); + struct xpsb_reloc one[XPSB_MAX_TA_RELOCS], two[XPSB_MAX_TA_RELOCS]; + struct xpsb_reloc_cfg c; + unsigned i, n, dsize = 0, diff = 0; + float consts[XPSB_VIDSHADER_SA_PACKED]; + uint32_t argb = 0xff336699; + + puts("scalar operand in sa0:"); + memset(prog, 0xee, sizeof prog); + n = xpsb_gen_sec_pds(prog, (const float *)&argb, 1, 0, &dsize); + expect("one constant, program length", n, 10); + expect("one constant, data segment", dsize, 8); + expect("the constant is data dword 0", prog[0], argb); + expect("attribute offset zero, so sa0", prog[1], 0); + for (i = 2; i < 8; i++) + if (prog[i]) diff++; + expect("rest of the data segment zero", diff, 0); + expect("code is DOUTA", prog[8], douta); + expect("code ends in HALT", prog[9], halt); + + /* the eleven-float packed video route is the same emitter's DMA case */ + for (i = 0; i < XPSB_VIDSHADER_SA_PACKED; i++) + consts[i] = (float)i; + n = xpsb_gen_sec_pds(prog, consts, XPSB_VIDSHADER_SA_PACKED, 0x20014800, + &dsize); + expect("eleven constants, program length", n, 14); + expect("eleven constants, data segment", dsize, 12); + expect("constant block address", prog[0], 0x20014800u); + expect("DMA control", prog[1], 0x80000000u | (10u << 21) | 10u); + expect("code is DOUTD", prog[12], doutd); + expect("code ends in HALT", prog[13], halt); + expect("zero constants refused", xpsb_gen_sec_pds(prog, NULL, 0, 0, NULL), + 0); + + xpsb_gen_heap(a, 128, 128, 0, 0); + xpsb_gen_heap(b, 128, 128, 0, 0); + /* the block the program moves into is free in drmcube's heap too */ + gen_heap(b, 128, 128, 0); + for (i = 0; i < XPSB_SEC_PDS_MAX; i++) + if (b[XPSB_SEC_PDS_OFF / 4 + i]) diff++; + expect("reference heap is zero there", diff, 0); + memset(b, 0, HEAP_DW * 4); + xpsb_gen_heap(b, 128, 128, 0, 0); + + expect("captured binding, address", a[PDS_SEC(a)], 0x00001038u); + expect("captured binding, no attributes", + a[PDS_CTL(a)] & XPSB_SA_BLOCK_MASK, 0); + + xpsb_heap_set_sec_pds(b, XPSB_SEC_PDS_OFF, 8, 1); + expect("binding address and data size", b[PDS_SEC(b)], + 0x08000000u | (((XPSB_HEAP_ADDR + XPSB_SEC_PDS_OFF) >> 4) & + 0x00ffffffu)); + expect("one block of secondary attributes", + b[PDS_CTL(b)] & XPSB_SA_BLOCK_MASK, 1u << 18); + diff = 0; + for (i = 0; i < HEAP_DW; i++) + if (a[i] != b[i] && i != PDS_SEC(a) && + i != PDS_CTL(a)) { + printf(" FAIL binding moved %#06x\n", i); + diff++; + } + expect("only the two binding words move", diff, 0); + xpsb_heap_set_sec_pds(b, 0, 0, 0); + expect_same("no constant restores the captured binding", a, b, HEAP_DW); + + /* the program's own 40 bytes land past the index buffer and touch + * nothing the captured heap describes */ + memcpy(b + XPSB_SEC_PDS_OFF / 4, prog, 10 * 4); + xpsb_heap_set_sec_pds(b, XPSB_SEC_PDS_OFF, 8, 1); + diff = 0; + for (i = 0; i < HEAP_DW; i++) + if (a[i] != b[i] && i != PDS_SEC(a) && + i != PDS_CTL(a) && + (i < XPSB_SEC_PDS_OFF / 4 || + i >= XPSB_SEC_PDS_OFF / 4 + XPSB_SEC_PDS_MAX)) + diff++; + expect("program lands in free heap", diff, 0); + + memset(&c, 0, sizeof c); + c.ntex = 1; + expect("default table length", xpsb_gen_ta_relocs(one, &c), + XPSB_NUM_TA_RELOCS); + c.sec_pds_off = XPSB_SEC_PDS_OFF; + c.sec_pds_dwords = 8; + expect("table length unchanged", xpsb_gen_ta_relocs(two, &c), + XPSB_NUM_TA_RELOCS); + /* nothing else in either table names the block it moved into */ + diff = 0; + for (i = 0; i < XPSB_NUM_TA_RELOCS; i++) + if (one[i].buffer == XPSB_BUF_HEAP && + one[i].pre_add >= XPSB_SEC_PDS_OFF && + one[i].pre_add < XPSB_SEC_PDS_OFF + 4 * XPSB_SEC_PDS_MAX) + diff++; + expect("the block is unclaimed", diff, 0); + expect("relocation pre_add", two[67].pre_add, XPSB_SEC_PDS_OFF); + expect("relocation background", two[67].background, 8u << 24); + expect("relocation destination", two[67].where, 0x1149u); + diff = 0; + for (i = 0; i < XPSB_NUM_TA_RELOCS; i++) + if (i != 67 && memcmp(&one[i], &two[i], sizeof one[0])) diff++; + expect("no other relocation moves", diff, 0); + expect("placeholder matches relocation", b[PDS_SEC(b)], + two[67].background | + (((XPSB_HEAP_ADDR + two[67].pre_add) >> 4) & two[67].mask)); + + printf(" sa0 <- %#010x, %u-dword program at heap+%#05x, data %u\n", + argb, 10, XPSB_SEC_PDS_OFF, 8); + free(a); + free(b); +} + +/* Everything the packed YUV binding writes, dword by dword against values + * computed here rather than read out of a header, and then the three buffers + * it touches diffed whole against the same buffers without it. + * + * The three shader words and the temporary-register count are the constants + * with no capture of this configuration behind them, so they are stated as + * literals: the first hardware run suspects these and nothing else. */ +static void test_video_bind(void) +{ + static const float conv[XPSB_VID_CONST_N] = { + 1.164f, 0.0f, 1.596f, 1.164f, -0.392f, -0.813f, + 1.164f, 2.017f, 0.0f, -0.0625f, 1.0f + }; + static const uint32_t pa1[3] = { 0xa0400082, 0xa0214082, 0xa0028082 }; + const unsigned vid = XPSB_USSE_VID_OFF / 4, sec = XPSB_SEC_PDS_OFF / 4; + uint32_t *a = calloc(HEAP_DW, 4), *b = calloc(HEAP_DW, 4); + uint32_t *u = calloc(XPSB_USSE_DW, 4), *v = calloc(XPSB_USSE_DW, 4); + struct xpsb_reloc one[XPSB_MAX_TA_RELOCS], two[XPSB_MAX_TA_RELOCS]; + struct xpsb_reloc_cfg c; + uint32_t prog[XPSB_SEC_PDS_MAX]; + const uint32_t *blob; + unsigned i, n1, n2, len = 0, dsize = 0, diff = 0; + + puts("packed video binding:"); + + /* ---- the USSE slot ---- */ + blob = xpsb_vidshader(XPSB_FOURCC_YUY2, &len); + xpsb_gen_usse(u, XPSB_CLEAR_DEFAULT); + xpsb_gen_usse(v, XPSB_CLEAR_DEFAULT); + expect("program length", xpsb_usse_video(u, XPSB_FOURCC_YUY2), len); + expect("source register moved", + (uint32_t)xpsb_usse_video_pa(u, XPSB_FOURCC_YUY2, + XPSB_VID_SRC_PA), 0); + expect("unpack 0 reads pa1", u[vid + 0], pa1[0]); + expect("unpack 1 reads pa1", u[vid + 2], pa1[1]); + expect("unpack 2 reads pa1", u[vid + 4], pa1[2]); + /* each of the three moved by exactly the one bit that carries pa1, and + * every other word of the program is the blob's */ + for (i = 0; i < len; i++) { + uint32_t want = blob[i]; + + if (i == 0 || i == 2 || i == 4) + want |= 1u << XPSB_PCK_SRC1_SHIFT; + if (u[vid + i] != want) { + printf(" FAIL program dword %#04x: %#010x != %#010x\n", + i, u[vid + i], want); + diff++; + } + } + expect("one bit per unpack, nothing else", diff, 0); + expect("the moved bit is w0[13:7] = 1", + (pa1[0] ^ blob[0]) | (pa1[1] ^ blob[2]) | (pa1[2] ^ blob[4]), + 1u << XPSB_PCK_SRC1_SHIFT); + diff = 0; + for (i = 0; i < XPSB_USSE_DW; i++) + if (u[i] != v[i] && (i < vid || i >= vid + XPSB_USSE_VID_SLOT)) + diff++; + expect("nothing outside the video slot moves", diff, 0); + expect("a planar FOURCC has no packed program", + (uint32_t)xpsb_usse_video_pa(u, XPSB_FOURCC_NV12, 1), + (uint32_t)-1); + expect("a pa the field cannot hold", + (uint32_t)xpsb_usse_video_pa(u, XPSB_FOURCC_YUY2, 128), + (uint32_t)-1); + + /* ---- the heap ---- */ + xpsb_gen_heap(a, 128, 128, 0, 0); + xpsb_gen_heap(b, 128, 128, 0, 0); + expect("captured USE task word 1 is zero", a[XPSB_HEAP_USE_TEMPS], 0); + expect("captured fragment program at 0x100", + (a[XPSB_PRI_PDS_DW] >> 8) & 0x7ff, XPSB_USSE_FRAG_OFF / 16); + + xpsb_heap_set_video(b, conv); + xpsb_heap_set_frag_use(b, XPSB_USSE_VID_OFF); + n1 = xpsb_gen_sec_pds(prog, conv, XPSB_VID_CONST_N, + XPSB_HEAP_ADDR + XPSB_VID_CONST_OFF, &dsize); + expect("secondary program length", n1, 14); + expect("secondary data segment", dsize, 12); + memcpy(b + sec, prog, n1 * 4); + xpsb_heap_set_sec_pds(b, XPSB_SEC_PDS_OFF, dsize, XPSB_VID_CONST_N); + + expect("six temporaries in bits [31:27]", b[XPSB_HEAP_USE_TEMPS], + 6u << 27); + expect("which is heap dword 0xd1", XPSB_HEAP_USE_TEMPS, 0xd1); + expect("video program at 0x200", + (b[XPSB_PRI_PDS_DW] >> 8) & 0x7ff, XPSB_USSE_VID_OFF / 16); + expect("only that field of the task word moves", + (a[XPSB_PRI_PDS_DW] ^ b[XPSB_PRI_PDS_DW]) & ~0x0007ff00u, 0); + expect("the dependency background is untouched", + b[XPSB_PRI_PDS_DW] & 0x180000u, 0x180000u); + expect("the iterator word is untouched", b[0xd9], 0x0fc0aa00u); + + expect("DMA source is heap+0x440", b[sec], + XPSB_HEAP_ADDR + XPSB_VID_CONST_OFF); + expect("eleven dwords to attribute 0", b[sec + 1], + 0x80000000u | (10u << 21) | 10u); + expect("code is DOUTD", b[sec + 12], 0x07030223u); + expect("code ends in HALT", b[sec + 13], 0xaf000000u); + expect("the constants are the caller's floats verbatim", + (uint32_t)memcmp(b + XPSB_VID_CONST_DW, conv, sizeof conv), 0); + expect("and they are 64-byte aligned", XPSB_VID_CONST_OFF & 63, 0); + expect("binding address and data size", b[PDS_SEC(b)], + 0x0c000000u | (((XPSB_HEAP_ADDR + XPSB_SEC_PDS_OFF) >> 4) & + 0x00ffffffu)); + expect("one block of secondary attributes", + b[PDS_CTL(b)], 0x30070002u); + expect("the varying count is unchanged", + b[PDS_CTL(b)] & XPSB_VARYING_MASK, + a[PDS_CTL(a)] & XPSB_VARYING_MASK); + + diff = 0; + for (i = 0; i < HEAP_DW; i++) { + if (a[i] == b[i]) continue; + if (i == XPSB_PRI_PDS_DW || i == XPSB_HEAP_USE_TEMPS || + i == PDS_SEC(a) || i == PDS_CTL(a) || + (i >= sec && i < sec + XPSB_SEC_PDS_MAX) || + (i >= XPSB_VID_CONST_DW && + i < XPSB_VID_CONST_DW + XPSB_VID_CONST_N)) + continue; + printf(" FAIL heap dword %#06x moved\n", i); + diff++; + } + expect("no other heap dword moves", diff, 0); + + /* ---- the relocation table ---- */ + memset(&c, 0, sizeof c); + c.ntex = 1; + n1 = xpsb_gen_ta_relocs(one, &c); + expect("captured table length", n1, XPSB_NUM_TA_RELOCS); + expect("captured fragment USE offset", one[49].pre_add, + XPSB_USSE_FRAG_OFF); + expect("captured fragment USE size", one[49].arg0, 8); + + c.sec_pds_off = XPSB_SEC_PDS_OFF; + c.sec_pds_dwords = dsize; + c.sec_const_off = XPSB_VID_CONST_OFF; + c.frag_use_off = XPSB_USSE_VID_OFF; + c.frag_use_size = XPSB_USSE_VID_SLOT * 4; + n2 = xpsb_gen_ta_relocs(two, &c); + expect("one record added", n2, n1 + 1); + expect("the table still fits", n2 <= XPSB_MAX_TA_RELOCS, 1); + + for (i = 49; i < 52; i++) { + expect("fragment USE offset", two[i].pre_add, XPSB_USSE_VID_OFF); + expect("fragment USE size", two[i].arg0, 0x80); + expect("fragment USE where", two[i].where, XPSB_PRI_PDS_DW); + two[i].pre_add = one[i].pre_add; + two[i].arg0 = one[i].arg0; + expect("nothing else in that record moves", + memcmp(&two[i], &one[i], sizeof one[0]) != 0, 0); + two[i].pre_add = XPSB_USSE_VID_OFF; + two[i].arg0 = 0x80; + } + expect("the program fits the size it reserves", + len * 4 <= XPSB_USSE_VID_SLOT * 4, 1); + + expect("secondary binding pre_add", two[67].pre_add, XPSB_SEC_PDS_OFF); + expect("secondary binding background", two[67].background, 12u << 24); + expect("constant block where", two[68].where, (uint32_t)sec); + expect("constant block op", two[68].reloc_op, XPSB_RELOC_OP_OFFSET); + expect("constant block source buffer", two[68].buffer, XPSB_BUF_HEAP); + expect("constant block destination", two[68].dst_buffer, XPSB_BUF_HEAP); + expect("constant block whole dword", two[68].mask, 0xffffffffu); + expect("constant block unshifted", two[68].shift, 0); + expect("constant block pre_add", two[68].pre_add, XPSB_VID_CONST_OFF); + expect("constant block background", two[68].background, 0); + expect("placeholder matches relocation", b[two[68].where], + (XPSB_HEAP_ADDR + two[68].pre_add) & two[68].mask); + + diff = 0; + for (i = 68; i < n1; i++) + if (memcmp(&two[i + 1], &one[i], sizeof one[0])) diff++; + expect("every later record is the captured one, moved", diff, 0); + diff = 0; + for (i = 0; i < 49; i++) + if (memcmp(&two[i], &one[i], sizeof one[0])) diff++; + for (i = 52; i < 67; i++) + if (memcmp(&two[i], &one[i], sizeof one[0])) diff++; + expect("no other record moves", diff, 0); + /* nothing in the captured table names the block the constants sit in */ + diff = 0; + for (i = 0; i < n1; i++) + if (one[i].buffer == XPSB_BUF_HEAP && + one[i].pre_add >= XPSB_VID_CONST_OFF && + one[i].pre_add < XPSB_VID_CONST_OFF + 4 * XPSB_VID_CONST_N) + diff++; + expect("the constant block is unclaimed", diff, 0); + + printf(" %u-instruction program at USSE+%#06x reading pa%u, " + "%u floats at heap+%#05x\n", len / 2, XPSB_USSE_VID_OFF, + XPSB_VID_SRC_PA, XPSB_VID_CONST_N, XPSB_VID_CONST_OFF); + free(a); + free(b); + free(u); + free(v); +} + +/* The video half: the programs and both coefficient routes are generated and + * checked. test_video_bind() above is what binds the packed ones. */ +static void test_video_gen(void) +{ + static const uint32_t fourcc[5] = { + XPSB_FOURCC_YUY2, XPSB_FOURCC_UYVY, XPSB_FOURCC_NV12, + XPSB_FOURCC_YV12, XPSB_FOURCC_I420 + }; + static const uint32_t coeff_regs[10] = { + 0x0a9c, 0x0a84, 0x0a88, 0x0a8c, 0x0a90, + 0x0a94, 0x0a98, 0x0b20, 0x0b24, 0x0b28 + }; + uint32_t sgx[9] = { 1, 2, 3, 4, 5, 6, 7, 8, 9 }; + uint32_t *u = calloc(XPSB_USSE_DW, 4), *v = calloc(XPSB_USSE_DW, 4); + uint32_t base[64], with[64]; + unsigned i, k, n, nb, diff; + + puts("video programs and coefficients:"); + for (k = 0; k < 5; k++) { + const uint32_t *prog; + unsigned len = 0; + + xpsb_gen_usse(u, XPSB_CLEAR_DEFAULT); + xpsb_gen_usse(v, XPSB_CLEAR_DEFAULT); + prog = xpsb_vidshader(fourcc[k], &len); + n = xpsb_usse_video(u, fourcc[k]); + expect("program length", n, len); + expect_same("program words", u + XPSB_USSE_VID_OFF / 4, prog, n); + diff = 0; + for (i = 0; i < XPSB_USSE_DW; i++) + if (u[i] != v[i] && + (i < XPSB_USSE_VID_OFF / 4 || + i >= XPSB_USSE_VID_OFF / 4 + XPSB_USSE_VID_SLOT)) + diff++; + expect("only the video slot moves", diff, 0); + } + xpsb_gen_usse(u, XPSB_CLEAR_DEFAULT); + xpsb_gen_usse(v, XPSB_CLEAR_DEFAULT); + expect("unknown FOURCC refused", xpsb_usse_video(u, 0x42424242u), 0); + expect_same("unknown FOURCC writes nothing", u, v, XPSB_USSE_DW); + + nb = xpsb_gen_ta_raster_stream(base, 128, 128); + n = xpsb_gen_ta_raster_stream_coeffs(with, 128, 128, sgx); + expect("coefficients append ten pairs", n, nb + 20); + expect_same("the register list is unchanged", base, with, nb); + expect("group count register", with[nb], coeff_regs[0]); + expect("group count", with[nb + 1], 3); + for (i = 0; i < 9; i++) { + expect("coefficient register", with[nb + 2 + 2 * i], + coeff_regs[1 + i]); + expect("coefficient value", with[nb + 3 + 2 * i], sgx[i]); + } + expect("no coefficients appends nothing", + xpsb_gen_ta_raster_stream_coeffs(with, 128, 128, NULL), nb); + + printf(" 5 FOURCCs into slot %#06x, %u register pairs\n", + XPSB_USSE_VID_OFF, XPSB_VIDSHADER_COEFF_PAIRS); + free(u); + free(v); +} + +static void test_relocs(void) +{ + struct xpsb_reloc ta[XPSB_MAX_TA_RELOCS], ras[XPSB_NUM_RAS_RELOCS]; + uint32_t rec[MISC_DW]; + struct xpsb_reloc_cfg c; + unsigned i, n, nras, draws = 0, rastclear = 0; + + puts("generated relocations, against test/reloc_tables.h:"); + memset(&c, 0, sizeof c); + c.first_draw = 0; c.ntex = 1; + + n = xpsb_gen_ta_relocs(ta, &c); + nras = xpsb_gen_raster_relocs(ras, &c); + expect("ta reloc count", n, XPSB_NUM_TA_RELOCS); + expect("raster reloc count", nras, XPSB_NUM_RAS_RELOCS); + expect("ta relocs byte-identical", + memcmp(ta, ta_relocs, sizeof ta_relocs) != 0, 0); + expect("raster relocs byte-identical", + memcmp(ras, raster_relocs, sizeof ras) != 0, 0); + for (i = 0; i < XPSB_NUM_TA_RELOCS; i++) + if (memcmp(&ta[i], &ta_relocs[i], sizeof ta[0])) + printf(" FAIL ta reloc %u differs\n", i); + + /* Record and relocation come from one descriptor, so the placeholder + * each relocation overwrites must rebuild from its own pre_add. */ + memset(rec, 0, sizeof rec); + xpsb_gen_draw_records_from(rec, 0); + for (i = 0; i < n; i++) { + uint32_t v; + + if (ta[i].dst_buffer != XPSB_BUF_VTX || + ta[i].buffer != XPSB_BUF_HEAP) continue; + v = (XPSB_HEAP_ADDR + ta[i].pre_add); + if (ta[i].shift == 0x40000) v >>= 4; + expect("draw placeholder matches reloc", + rec[ta[i].where], ta[i].background | (v & ta[i].mask)); + draws++; + } + expect("draw-record relocations", draws, 10); + + c.first_draw = XPSB_USER_DRAW; + n = xpsb_gen_ta_relocs(ta, &c); + expect("no-clear ta reloc count", n, XPSB_NUM_TA_RELOCS - 4 - 6); + for (i = 0; i < n; i++) + if (ta[i].dst_buffer == XPSB_BUF_RASTGEOM) rastclear++; + expect("clearing rastgeom relocations gone", rastclear, 0); + memset(rec, 0, sizeof rec); + xpsb_gen_draw_records_from(rec, XPSB_USER_DRAW); + draws = 0; + for (i = 0; i < n; i++) { + uint32_t v; + + if (ta[i].dst_buffer != XPSB_BUF_VTX || + ta[i].buffer != XPSB_BUF_HEAP) continue; + v = (XPSB_HEAP_ADDR + ta[i].pre_add); + if (ta[i].shift == 0x40000) v >>= 4; + expect("no-clear placeholder matches reloc", + rec[ta[i].where], ta[i].background | (v & ta[i].mask)); + expect("no-clear where in range", ta[i].where > 7, 0); + draws++; + } + expect("no-clear draw relocations", draws, 4); + + c.dest_offset = 0x2000; + c.tex_offset[0] = 0x800; + c.first_draw = 0; + n = xpsb_gen_ta_relocs(ta, &c); + nras = xpsb_gen_raster_relocs(ras, &c); + expect("dest offset at heap[1]", ta[0].pre_add, 0x2000); + expect("dest offset at heap[0x52]", ta[16].pre_add, 0x2000); + expect("texture offset", ta[48].pre_add, 0x800); + expect("texture where", ta[48].where, 0xda); + expect("raster blit source offset", ras[11].pre_add, 0x2000); + + /* two units is step 7, checked in detail in test_two_textures() */ + c.ntex = 2; + expect("two texture units", xpsb_gen_ta_relocs(ta, &c), + XPSB_NUM_TA_RELOCS + 1); + c.ntex = 3; + expect("three texture units refused", xpsb_gen_ta_relocs(ta, &c), 0); + c.ntex = 1; c.first_draw = 1; + expect("half a clear refused", xpsb_gen_ta_relocs(ta, &c), 0); + printf(" %u+%u regenerated, %u+%u without the clear\n", + XPSB_NUM_TA_RELOCS, XPSB_NUM_RAS_RELOCS, + XPSB_NUM_TA_RELOCS - 10, XPSB_NUM_RAS_RELOCS); +} + +int main(void) +{ + test_heap(); + test_heap_nonsquare(); + test_heap_stride(); + test_rastgeom(); + test_streams(); + test_msaa(); + test_tables(); + test_texstate(); + test_stride_refusal(); + test_dest_heap(); + test_quad(); + test_noclear_records(); + test_relocs(); + test_usse(); + test_cull(); + test_state_layout(); + test_state_carry(); + test_blend(); + test_two_textures(); + test_scalar(); + test_video_gen(); + test_video_bind(); + + printf("\n%s: %d failure%s\n", failures ? "FAILED" : "ok", failures, + failures == 1 ? "" : "s"); + return failures != 0; +} diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/test_heap.c mesa-26.2.2/src/gallium/drivers/sgx/test_heap.c --- mesa-26.2.2.orig/src/gallium/drivers/sgx/test_heap.c 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/test_heap.c 2026-09-08 10:57:36.684553420 +0200 @@ -0,0 +1,58 @@ +/* Synthesis has to reproduce the captured parameter heap exactly. + * + * This is the gate on replacing the transcription with code: a category is + * either generated and bit-identical to the capture, or it is still a literal. + * Nothing in between, and the count of what is still literal is printed so the + * migration has a number attached to it. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: GPL-2.0-only + */ +#include "xpsb_heap.h" + +#include +#include +#include + +#define HEAP_DW 0x10800u + +int main(void) +{ + uint32_t *synth = calloc(HEAP_DW, 4); + uint32_t *cap = calloc(HEAP_DW, 4); + unsigned i, bad = 0, gen, lit; + + if (!synth || !cap) + return 77; + + for (i = 0; i < xpsb_heap_captured_n; i++) + cap[xpsb_heap_captured[i].at] = xpsb_heap_captured[i].v; + xpsb_heap_synth(synth, NULL); + + for (i = 0; i < HEAP_DW; i++) { + if (synth[i] == cap[i]) + continue; + if (bad < 12) + printf(" FAIL heap dword 0x%05x: synthesised 0x%08x, " + "captured 0x%08x\n", i, synth[i], cap[i]); + bad++; + } + + gen = xpsb_heap_synthesised(); + lit = xpsb_heap_literals(); + printf("parameter heap: %u dword(s) generated, %u still literal, " + "%u total\n", gen, lit, gen + lit); + printf(" %.0f%% synthesised\n", 100.0 * gen / (gen + lit)); + if (gen + lit != xpsb_heap_captured_n) { + printf("FAIL: %u accounted for, capture has %u\n", + gen + lit, xpsb_heap_captured_n); + bad++; + } + if (bad) + printf("FAILED: %u dword(s) differ\n", bad); + else + printf("ok: synthesis reproduces the capture exactly\n"); + free(synth); free(cap); + return bad ? 1 : 0; +} diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/test_hw.c mesa-26.2.2/src/gallium/drivers/sgx/test_hw.c --- mesa-26.2.2.orig/src/gallium/drivers/sgx/test_hw.c 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/test_hw.c 2026-09-08 10:57:36.684562021 +0200 @@ -0,0 +1,314 @@ +/* Self-test for the reversed Xpsb hardware layer. + * + * The register file is replaced by ordinary memory, so every handler can be + * driven without hardware: the scene and TA-memory arithmetic is checked + * against values read out of the disassembly, and the register writes each + * fire path performs are printed for comparison against a capture. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#include +#include + +#include "xpsb_hw.h" +#include "xpsb_shader.h" + +static uint8_t regfile[0x1000]; +static int failures; + +static void expect(const char *what, uint32_t got, uint32_t want) +{ + if (got != want) { + printf(" FAIL %-24s got %#010x want %#010x\n", what, got, want); + failures++; + } +} + +/* The blob rejects nothing; past 2047 pixels it switches to 4x4 macro tiles + * with a three-lane packing. Both branches are exercised here. */ +static void test_scene_info(void) +{ + static const struct { + uint32_t dim, tiles, mtile, packed; + } cases[] = { + { 128, 8, 4, 0x01004004 }, + { 256, 16, 8, 0x02008008 }, + { 304, 19, 12, 0x0300c00c }, + { 512, 32, 16, 0x04010010 }, + { 640, 40, 20, 0x05014014 }, + { 2047, 128, 64, 0x10040040 }, + }; + struct xpsb_hw hw; + uint32_t cookie[16], size, cps, cnp; + unsigned i; + + memset(&hw, 0, sizeof hw); + hw.reg = regfile; + + puts("scene_info, 2x2 macro tiles:"); + for (i = 0; i < sizeof cases / sizeof cases[0]; i++) { + uint32_t d = cases[i].dim; + + memset(cookie, 0xa5, sizeof cookie); + xpsb_scene_info(&hw, d, d, cookie, &size, &cps, &cnp); + printf(" %4u: tiles %ux%u mtile %u packed %#010x " + "size %u clear %u+%u\n", d, (d + 15) >> 4, (d + 15) >> 4, + cases[i].mtile, cookie[2], size, cps, cnp); + expect("packed_x", cookie[2], cases[i].packed); + expect("packed_y", cookie[3], cases[i].packed); + expect("mtile count", cookie[1], + cases[i].mtile * cases[i].mtile); + expect("tile extent", cookie[5], + ((cases[i].tiles - 1) << 12) | (cases[i].tiles - 1)); + expect("pixel extent", cookie[6], ((d - 1) << 12) | (d - 1)); + expect("clear start", cps, 0); + expect("clear pages", cnp, (cookie[7] + 0xfff) >> 12); + } + + puts("scene_info, 4x4 macro tiles past 2047:"); + memset(cookie, 0, sizeof cookie); + xpsb_scene_info(&hw, 2048, 2048, cookie, &size, &cps, &cnp); + printf(" 2048: mode %#010x packed %#010x size %u\n", + cookie[0], cookie[2], size); + expect("large mode", cookie[0], 0x80000000); + /* tiles 128 -> ceil(128/4) = 32, aligned to 4 */ + expect("large packed_x", cookie[2], + 0x80000000u | (32u << 22) | (32u << 13) | (32u * 3u)); + + puts("scene_info, large-scene workaround keeps the 2x2 rule:"); + hw.vopt[VOPT_LARGE_SCENE] = 1; + memset(cookie, 0, sizeof cookie); + xpsb_scene_info(&hw, 2048, 2048, cookie, &size, &cps, &cnp); + expect("workaround mode", cookie[0], 0); + expect("workaround packed_x", cookie[2], + (64u << 22) | (64u << 12) | 64u); +} + +static void test_ta_mem_info(void) +{ + uint32_t cookie[16], size; + int ret; + + puts("ta_mem_info:"); + memset(cookie, 0, sizeof cookie); + ret = xpsb_ta_mem_info(0x61f, cookie, &size); + printf(" 0x61f pages -> ret %d size %#x\n", ret, size); + expect("too small rejected", (uint32_t)ret, (uint32_t)-12); + expect("size reported anyway", size, 0x620000); + + ret = xpsb_ta_mem_info(0x800, cookie, &size); + printf(" 0x800 pages -> ret %d size %#x avail %#x tail %#x\n", + ret, size, cookie[3], cookie[6]); + expect("accepted", (uint32_t)ret, 0); + expect("pages", cookie[2], 0x800); + expect("avail", cookie[3], 0x200); + expect("tail", cookie[6], 0x1c0); + + ret = xpsb_ta_mem_info(0x620, cookie, &size); + printf(" 0x620 pages -> avail %#x tail %#x\n", cookie[3], cookie[6]); + expect("minimum accepted", (uint32_t)ret, 0); + expect("tail clamped", cookie[6], 0); +} + +static void test_vopt(void) +{ + struct xpsb_hw hw; + + puts("set_vopt:"); + memset(&hw, 0, sizeof hw); + hw.reg = regfile; + + /* SGX535 as fitted to Poulsbo: core revision 1.2.1 -> key 121. */ + *(uint32_t *)(regfile + SGX_CORE_REVISION) = 0x00010201; + xpsb_set_vopt(&hw); + printf(" rev 1.2.1 -> large_scene %u wide_mtile %u dpm_limit %u\n", + hw.vopt[VOPT_LARGE_SCENE], hw.vopt[VOPT_WIDE_MTILE], + hw.vopt[VOPT_DPM_LIMIT]); + expect("no workarounds", hw.vopt[VOPT_LARGE_SCENE] | + hw.vopt[VOPT_WIDE_MTILE] | hw.vopt[VOPT_DPM_LIMIT], 0); + + memset(&hw, 0, sizeof hw); + hw.reg = regfile; + *(uint32_t *)(regfile + SGX_CORE_REVISION) = 0x00010009; + xpsb_set_vopt(&hw); + printf(" rev 1.0.9 -> all workarounds %u\n", hw.vopt[VOPT_DPM_LIMIT]); + expect("rev 109 enables everything", hw.vopt[VOPT_DPM_LIMIT], 1); +} + +static void test_sgx_initialize(void) +{ + struct xpsb_hw hw; + + puts("sgx_initialize:"); + memset(&hw, 0, sizeof hw); + memset(regfile, 0, sizeof regfile); + hw.reg = regfile; + xpsb_sgx_initialize(&hw); + printf(" dpm limit %#010x use_ctrl %#010x tsp %#010x\n", + *(uint32_t *)(regfile + SGX_DPM_LIMIT_A), hw.use_ctrl, + *(uint32_t *)(regfile + SGX_TSP_CTRL)); + expect("dpm limit", *(uint32_t *)(regfile + SGX_DPM_LIMIT_A), + 0x5188200); + expect("use ctrl", *(uint32_t *)(regfile + SGX_USE_CTRL), 0xffff); + expect("bif ctrl", *(uint32_t *)(regfile + SGX_BIF_CTRL2), 0xc07c); + + hw.vopt[VOPT_CLKGATE] = 1; + hw.vopt[VOPT_DPM_LIMIT] = 1; + xpsb_sgx_initialize(&hw); + expect("gated use ctrl", *(uint32_t *)(regfile + SGX_USE_CTRL), + 0x100ffff); + expect("halved dpm limit", *(uint32_t *)(regfile + SGX_DPM_LIMIT_A), + 0x5021900); +} + +static void test_oom_abort(void) +{ + struct xpsb_hw hw; + uint32_t cookie[16], bca, rca, flags; + + puts("oom_abort:"); + memset(&hw, 0, sizeof hw); + memset(regfile, 0, sizeof regfile); + hw.reg = regfile; + memset(cookie, 0, sizeof cookie); + + *(uint32_t *)(regfile + SGX_DPM_STATUS) = 0; + xpsb_oom_abort(&hw, cookie, &bca, &rca, &flags); + printf(" clean: mode %u bca %u rca %u flags %#x\n", + *(uint32_t *)(regfile + SGX_DPM_MODE), bca, rca, flags); + expect("mode", *(uint32_t *)(regfile + SGX_DPM_MODE), 2); + expect("bca = RASTER_BLOCK", bca, 0); + expect("rca = TA", rca, 3); + expect("flags = XHW_OOM", flags, 0x1000000); + expect("pass marker", cookie[14], 1); + + cookie[13] = 0x40000000; + xpsb_oom_abort(&hw, cookie, &bca, &rca, &flags); + printf(" retry: mode %u cookie15 %#x\n", + *(uint32_t *)(regfile + SGX_DPM_MODE), cookie[15]); + expect("retry mode", *(uint32_t *)(regfile + SGX_DPM_MODE), 4); + expect("forced status", cookie[15], 0x10); +} + +/* A clean TA fire with no hardware behind the registers: the kick helpers + * will time out, which is fine, what is being checked is the register + * programming order. */ +static void test_scene_fire(void) +{ + struct xpsb_hw hw; + uint32_t cookie[16], size, cps, cnp, rca = 0; + + puts("scene_switch_fire, clean TA bind:"); + memset(&hw, 0, sizeof hw); + memset(regfile, 0, sizeof regfile); + hw.reg = regfile; + xpsb_scene_info(&hw, 640, 640, cookie, &size, &cps, &cnp); + + /* Pre-set the status bits the kicks poll for so they return at once. */ + *(uint32_t *)(regfile + SGX_EVENT_STATUS) = 0xffffffff; + *(uint32_t *)(regfile + SGX_EVENT_STATUS3) = 0; + + xpsb_scene_switch_fire(&hw, 0, 0, cookie, 0x400000, 0, + 4 /* SETUP */, NULL, 0, &rca); + printf(" param base %#010x region base %#010x mtile %#010x/%#010x\n", + *(uint32_t *)(regfile + SGX_TA_PARAM_BASE), + *(uint32_t *)(regfile + SGX_TA_REGION_BASE), + *(uint32_t *)(regfile + SGX_TA_MTILE_X), + *(uint32_t *)(regfile + SGX_TA_MTILE_Y)); + expect("param base", *(uint32_t *)(regfile + SGX_TA_PARAM_BASE), + 0x400000 + cookie[8]); + expect("region base", *(uint32_t *)(regfile + SGX_TA_REGION_BASE), + 0x400000 + cookie[7]); + expect("mtile x", *(uint32_t *)(regfile + SGX_TA_MTILE_X), cookie[2]); + expect("mtile stride", *(uint32_t *)(regfile + SGX_TA_MTILE_STRIDE), + cookie[4]); + expect("pixel extent", *(uint32_t *)(regfile + SGX_TA_PIXEL_EXTENT), + cookie[6]); + expect("ta kicked", *(uint32_t *)(regfile + SGX_TA_KICK), 1); +} + + +/* The composite shader is two SOP2 words per operator. The blend halves come + * straight from the blob, so what is worth checking is the modulate half the + * binary builds from the picture formats, and the two operator overrides. */ +static void test_composite_shader(void) +{ + static const struct { + const char *name; + int op; + struct xpsb_shader_flags f; + uint32_t w0, w1; + } cases[] = { + { "tex src with alpha", XPSB_OP_OVER, {0,0,0,0}, + 0xa0000001, 0x80e80103 }, + { "tex src no alpha", XPSB_OP_OVER, {1,0,0,0}, + 0xa0000001, 0x80c80107 }, + { "a8 src", XPSB_OP_OVER, {0,1,0,0}, + 0xa0000001, 0x80e80003 }, + { "a8 src, scalar mask", XPSB_OP_OVER, {0,1,0,1}, + 0xb0000000, 0x80e80003 }, + { "scalar mask", XPSB_OP_OVER, {0,0,0,1}, + 0xb0000000, 0x80e80103 }, + { "scalar src", XPSB_OP_OVER, {0,0,1,0}, + 0xe0000000, 0x80e80103 }, + { "op Src ends early", XPSB_OP_SRC, {0,0,0,0}, + 0xa0000001, 0x80a40101 }, + { "op Saturate min", XPSB_OP_SATURATE, {0,0,0,0}, + 0xa0000001, 0x90e80003 }, + }; + uint32_t w[4]; + unsigned i; + int op; + + puts("composite_shader, modulate half:"); + for (i = 0; i < sizeof cases / sizeof cases[0]; i++) { + if (xpsb_composite_shader(cases[i].op, &cases[i].f, w)) { + printf(" FAIL %s rejected\n", cases[i].name); + failures++; + continue; + } + printf(" %-20s %#010x %#010x\n", cases[i].name, w[0], w[1]); + expect(cases[i].name, w[0], cases[i].w0); + expect(cases[i].name, w[1], cases[i].w1); + } + + puts("composite_shader, blend half present for every operator:"); + for (op = 0; op < XPSB_NUM_OPS; op++) { + struct xpsb_shader_flags f = {0, 0, 0, 0}; + + if (xpsb_composite_shader(op, &f, w)) { + printf(" FAIL op %d rejected\n", op); + failures++; + continue; + } + if (op != XPSB_OP_CLEAR && w[2] != 0xd0000000) { + printf(" FAIL op %d blend lo %#010x\n", op, w[2]); + failures++; + } + } + expect("Clear is a mov", 0, xpsb_composite_shader(XPSB_OP_CLEAR, + &(struct xpsb_shader_flags){0,0,0,0}, w)); + expect("Clear blend hi", w[3], 0xfca40001); + expect("out of range rejected", (uint32_t) + xpsb_composite_shader(14, &(struct xpsb_shader_flags){0,0,0,0}, w), + (uint32_t)-1); + expect("Clear opaque", xpsb_opaque_ops[XPSB_OP_CLEAR], 1); + expect("Over not opaque", xpsb_opaque_ops[XPSB_OP_OVER], 0); +} + +int main(void) +{ + test_scene_info(); + test_ta_mem_info(); + test_vopt(); + test_sgx_initialize(); + test_oom_abort(); + test_scene_fire(); + test_composite_shader(); + + printf("\n%s: %d failure%s\n", failures ? "FAILED" : "ok", failures, + failures == 1 ? "" : "s"); + return failures != 0; +} diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/test_pds.c mesa-26.2.2/src/gallium/drivers/sgx/test_pds.c --- mesa-26.2.2.orig/src/gallium/drivers/sgx/test_pds.c 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/test_pds.c 2026-09-08 10:57:36.684571441 +0200 @@ -0,0 +1,274 @@ +/* Open replacement for Xpsb.so - primary PDS program generator tests. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + * + * Every expectation here is a capture, not a derivation: + * + * stream the one-unit program the working stream carries at heap+0x340, + * dwords 0xd0..0xdf of xpsb_frame.c's heap_init[] + * P_mt2 tools/isa-pds/pds_corpus.txt:707, data@20010340 n=12 code@20010370 + * P_mt3 tools/isa-pds/pds_corpus.txt:818, data@20010340 n=16 code@20010380 + * + * A mismatch prints the differing dword and exits non-zero. + */ +#include "xpsb_pds.h" + +#include +#include + +static int fails; + +static int cmp(const char *what, const uint32_t *got, const uint32_t *want, + unsigned n) +{ + unsigned i; + int bad = 0; + + for (i = 0; i < n; i++) + if (got[i] != want[i]) { + printf(" FAIL %s[%u]: got %08x want %08x\n", what, i, + got[i], want[i]); + bad = 1; + } + fails += bad; + return bad; +} + +static int check(const char *name, unsigned ntex, + const struct xpsb_pds_tex *tex, const uint32_t use[3], + const uint32_t *want_data, unsigned want_ndata, + const uint32_t *want_code, unsigned want_ncode) +{ + uint32_t data[XPSB_PDS_MAX_DATA], code[XPSB_PDS_MAX_CODE]; + unsigned nd, nc, off; + int bad = 0; + + if (xpsb_pds_gen_primary(ntex, tex, use, data, &nd, code, &nc, &off)) { + printf(" FAIL %s: generator refused ntex=%u\n", name, ntex); + fails++; + return 1; + } + if (nd != want_ndata || nc != want_ncode || off != want_ndata * 4) { + printf(" FAIL %s: ndata %u/%u ncode %u/%u code_off %u/%u\n", + name, nd, want_ndata, nc, want_ncode, off, + want_ndata * 4); + fails++; + bad = 1; + } + bad |= cmp("data", data, want_data, want_ndata); + bad |= cmp("code", code, want_code, want_ncode); + printf(" %-24s ntex=%u data=%u dwords code=%u dwords code_off=0x%02x %s\n", + name, ntex, nd, nc, off, bad ? "MISMATCH" : "byte-identical"); + return bad; +} + +int main(void) +{ + /* xpsb_frame.c heap_init[], dwords 0xd0..0xdb and 0xdc..0xdf */ + static const struct xpsb_pds_tex stream_tex[1] = { + { 0x03fe0090, 0x6c03f03f, 0x804ea000, 0x0fc0aa00 }, + }; + static const uint32_t stream_use[3] = { + 0x00181025, 0x00000000, 0x00000020, + }; + static const uint32_t stream_data[12] = { + 0x00181025, 0x00000000, 0x03fe0090, 0x6c03f03f, + 0x00000000, 0x00000000, 0x00000000, 0x00000000, + 0x00000020, 0x0fc0aa00, 0x804ea000, 0x00000000, + }; + static const uint32_t stream_code[4] = { + 0x07000345, 0x070418a2, 0x07042364, 0xaf000000, + }; + + /* P_mt2 */ + static const struct xpsb_pds_tex mt2_tex[1] = { + { 0x03fe0090, 0x6c01f01f, 0x804ee000, 0x0fc0aa01 }, + }; + static const uint32_t mt2_use[3] = { + 0x0018102c, 0x00000000, 0x00000020, + }; + static const uint32_t mt2_data[12] = { + 0x0018102c, 0x00000000, 0x03fe0090, 0x6c01f01f, + 0x00000000, 0x00000000, 0x00000000, 0x00000000, + 0x00000020, 0x0fc0aa01, 0x804ee000, 0x00000000, + }; + static const uint32_t mt2_code[4] = { + 0x07000345, 0x070418a2, 0x07042364, 0xaf000000, + }; + + /* P_mt3 - the only two-unit primary program in the corpus */ + static const struct xpsb_pds_tex mt3_tex[2] = { + { 0x03fe0090, 0x6c01f01f, 0x804ee000, 0x0fc0a201 }, + { 0x03fe0090, 0x6c00f00f, 0x804ef000, 0x0c00fa02 }, + }; + static const uint32_t mt3_use[3] = { + 0x0018102c, 0x00000000, 0x00000020, + }; + static const uint32_t mt3_data[16] = { + 0x0018102c, 0x00000000, 0x03fe0090, 0x6c01f01f, + 0x03fe0090, 0x6c00f00f, 0x00000000, 0x00000000, + 0x00000020, 0x0fc0a201, 0x804ee000, 0x0c00fa02, + 0x804ef000, 0x00000000, 0x00000000, 0x00000000, + }; + static const uint32_t mt3_code[6] = { + 0x07000345, 0x070418a2, 0x07042364, 0x070438a2, + 0x07084364, 0xaf000000, + }; + + uint32_t data[XPSB_PDS_MAX_DATA], code[XPSB_PDS_MAX_CODE]; + unsigned nd, nc, off; + + puts("primary PDS program vs captures:"); + check("stream heap+0x340", 1, stream_tex, stream_use, + stream_data, 12, stream_code, 4); + check("corpus P_mt2", 1, mt2_tex, mt2_use, mt2_data, 12, mt2_code, 4); + check("corpus P_mt3", 2, mt3_tex, mt3_use, mt3_data, 16, mt3_code, 6); + + if (!xpsb_pds_gen_primary(0, mt3_tex, mt3_use, data, &nd, code, &nc, &off) || + !xpsb_pds_gen_primary(XPSB_PDS_NTEX_MAX + 1, mt3_tex, mt3_use, data, + &nd, code, &nc, &off)) { + puts(" FAIL: out-of-range ntex accepted"); + fails++; + } + + /* The second DOUTT is the word this generator exists for. */ + xpsb_pds_gen_primary(2, mt3_tex, mt3_use, data, &nd, code, &nc, &off); + if (code[4] != 0x07084364) { + printf(" FAIL: second doutt %08x, capture has 07084364\n", + code[4]); + fails++; + } + + /* Four issues reach memory dword 24 of a sixteen-dword data array: + * refused, with nothing written past either array. */ + { + struct xpsb_pds_issue four[XPSB_PDS_MAX_ISSUE]; + uint32_t guarded[XPSB_PDS_MAX_DATA + XPSB_PDS_MAX_CODE + 12]; + unsigned i, spilt = 0; + + memset(four, 0, sizeof four); + for (i = 0; i < XPSB_PDS_MAX_ISSUE; i++) { + four[i].texissue = XPSB_DOUTI_NONE; + four[i].useissue = (uint8_t)i; + four[i].usedim = 4; + four[i].unpacked = 1; + } + memset(guarded, 0xa5, sizeof guarded); + /* Four issues that sample nothing fit: a DOUTI reads only + * ds1[1 + 2i], so the highest they reach is ds1[7], memory + * dword fifteen of sixteen. The slots at 2 + 2i belong to the + * DOUTT, and reserving them for issues with no texture is what + * used to refuse this. */ + if (xpsb_pds_gen_primary_issues(4, four, mt3_use, guarded, &nd, + guarded + XPSB_PDS_MAX_DATA + + 12, &nc, &off)) { + puts(" FAIL: four iterators refused"); + fails++; + } + /* Give them textures and the captured slot refuses them + * again: the last one reaches ds1[8], memory dword 24 of the + * sixteen it holds. */ + for (i = 0; i < XPSB_PDS_MAX_ISSUE; i++) + four[i].texissue = (uint8_t)i; + if (!xpsb_pds_gen_primary_issues(4, four, mt3_use, guarded, + &nd, guarded + + XPSB_PDS_MAX_DATA + 12, + &nc, &off)) { + puts(" FAIL: four sampling issues accepted in the " + "captured slot"); + fails++; + } + /* The spare block takes them, and a full list besides: eight + * issues reach memory dword 40. That is the whole point of + * moving the program - the ceiling was the region, not the + * part. */ + if (xpsb_pds_gen_primary_issues_max(XPSB_PDS_MAX_ISSUE, four, + mt3_use, guarded, &nd, + guarded + + XPSB_PDS_MAX_DATA + 12, + &nc, &off, + XPSB_PDS_MAX_DATA)) { + puts(" FAIL: a full list of samplers refused"); + fails++; + } + /* One more than the array holds is still refused. */ + if (!xpsb_pds_gen_primary_issues_max(XPSB_PDS_MAX_ISSUE + 1, + four, mt3_use, guarded, + &nd, guarded + + XPSB_PDS_MAX_DATA + 12, + &nc, &off, + XPSB_PDS_MAX_DATA)) { + puts(" FAIL: a list past the array accepted"); + fails++; + } + for (i = XPSB_PDS_MAX_DATA; i < XPSB_PDS_MAX_DATA + 12; i++) + if (guarded[i] != 0xa5a5a5a5u) + spilt++; + if (spilt) { + printf(" FAIL: %u dword(s) written past the data " + "array\n", spilt); + fails++; + } + xpsb_pds_gen_primary(1, mt2_tex, mt2_use, data, &nd, code, &nc, + &off); + printf(" %-24s four iterators fit the captured slot and four " + "samplers do not, the spare block takes %u, nothing " + "spilt\n", "issue ceiling", XPSB_PDS_MAX_ISSUE); + } + + /* An iterated set delivered as half floats. Nothing here is a capture: + * no captured frame carries the format, so both expectations are the + * DDK's - DOUTI[29:28] is EURASIA_PDS_DOUTI_USEFORMAT_F16 + * (sgxdefs.h:3760), and USEDIM still counts components while the + * registers the issue claims halve (usp_inputdata.c:1401-1409). */ + { + struct xpsb_pds_issue f16[2]; + uint32_t w; + + memset(f16, 0, sizeof f16); + f16[0].texissue = XPSB_DOUTI_NONE; + f16[0].useissue = 0; + f16[0].usedim = 4; + f16[0].unpacked = 1; + f16[0].usef16 = 1; + f16[1] = f16[0]; + f16[1].useissue = 1; + f16[1].usedim = 3; + if (xpsb_pds_gen_primary_issues(2, f16, mt3_use, data, &nd, + code, &nc, &off)) { + puts(" FAIL: an f16 issue list was refused"); + fails++; + } + w = data[xpsb_pds_ds_dword(1, 1)]; /* issue 0's word */ + if ((w >> 28) != 2u) { + printf(" FAIL: f16 douti %08x has useformat %u, " + "want 2\n", w, w >> 28); + fails++; + } + if (((w >> 22) & 3u) != 3u) { + printf(" FAIL: f16 douti %08x has usedim %u, want 4 " + "components\n", w, ((w >> 22) & 3u) + 1u); + fails++; + } + /* Four components in two registers, three in two: four. */ + if (xpsb_pds_issue_regs(2, f16) != 4) { + printf(" FAIL: two f16 sets claim %u register(s), " + "want 4\n", xpsb_pds_issue_regs(2, f16)); + fails++; + } + f16[0].usef16 = f16[1].usef16 = 0; + if (xpsb_pds_issue_regs(2, f16) != 7) { + printf(" FAIL: the same sets as f32 claim %u " + "register(s), want 7\n", + xpsb_pds_issue_regs(2, f16)); + fails++; + } + printf(" %-24s useformat 2, usedim in components, 4 " + "registers against 7\n", "f16 iteration"); + } + + printf("%s\n", fails ? "FAILED" : "all programs match the captures"); + return fails != 0; +} diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/test_sw.c mesa-26.2.2/src/gallium/drivers/sgx/test_sw.c --- mesa-26.2.2.orig/src/gallium/drivers/sgx/test_sw.c 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/test_sw.c 2026-09-08 10:57:36.684584112 +0200 @@ -0,0 +1,934 @@ +/* Self-test for the software pixel, composite and video paths. + * + * Nothing here touches hardware. Every expected value is recomputed in this + * file from the Render and BT.601 definitions instead of being taken from the + * module under test, so a wrong implementation cannot agree with a wrong + * expectation. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#include +#include + +#include "xpsb_comp.h" +#include "xpsb_pixel.h" +#include "xpsb_yuv.h" + +#define PICT_a8r8g8b8 0x20028888 +#define PICT_x8r8g8b8 0x20020888 +#define PICT_r5g6b5 0x10020565 +#define PICT_x1r5g5b5 0x10021555 +#define PICT_a8 0x08018000 +#define PICT_a8b8g8r8 0x20038888 + +static int failures; + +static void fail(const char *what) +{ + printf(" FAIL %s\n", what); + failures++; +} + +static void expect(const char *what, uint32_t got, uint32_t want) +{ + if (got != want) { + printf(" FAIL %-24s got %#010x want %#010x\n", what, got, want); + failures++; + } +} + +static void expect_near(const char *what, int got, int want, int tol) +{ + if (got < want - tol || got > want + tol) { + printf(" FAIL %-24s got %d want %d+-%d\n", what, got, want, tol); + failures++; + } +} + +/* The DDX bit-expansion rule, restated from psbPixelARGB8888. */ +static unsigned int ref_expand(unsigned int v, unsigned int bits) +{ + unsigned int hi = v << (8 - bits); + + return (v & 1) ? hi | ((1u << (8 - bits)) - 1u) : hi; +} + +static uint32_t ref_unpack565(uint16_t raw) +{ + return 0xff000000u | (ref_expand(raw >> 11, 5) << 16) | + (ref_expand((raw >> 5) & 0x3f, 6) << 8) | + ref_expand(raw & 0x1f, 5); +} + +static uint16_t ref_pack565(uint32_t argb) +{ + return (uint16_t)(((argb >> 19) & 0x1f) << 11 | + ((argb >> 10) & 0x3f) << 5 | + ((argb >> 3) & 0x1f)); +} + +static const struct { + const char *name; + unsigned int pict; + unsigned int bpp; + unsigned int bits[4]; + unsigned int shift[4]; +} known_formats[] = { + { "a8r8g8b8", PICT_a8r8g8b8, 32, { 8, 8, 8, 8 }, { 24, 16, 8, 0 } }, + { "x8r8g8b8", PICT_x8r8g8b8, 32, { 0, 8, 8, 8 }, { 24, 16, 8, 0 } }, + { "r5g6b5", PICT_r5g6b5, 16, { 0, 5, 6, 5 }, { 16, 11, 5, 0 } }, + { "x1r5g5b5", PICT_x1r5g5b5, 16, { 1, 5, 5, 5 }, { 15, 10, 5, 0 } }, + { "a8", PICT_a8, 8, { 8, 0, 0, 0 }, { 0, 0, 0, 0 } }, + { "a8b8g8r8", PICT_a8b8g8r8, 32, { 8, 8, 8, 8 }, { 24, 0, 8, 16 } }, +}; + +static int roundtrip(const struct xpsb_format *f, uint32_t raw) +{ + uint32_t out = 0; + + xpsb_pix_store(f, (uint8_t *)&out, + xpsb_pix_load(f, (const uint8_t *)&raw)); + return out == raw; +} + +static void test_format(void) +{ + static const uint32_t probe[8] = { 0, 1, 2, 3, 127, 128, 254, 255 }; + struct xpsb_format f; + char tag[64]; + unsigned int i, c, v; + + puts("format parse:"); + for (i = 0; i < sizeof known_formats / sizeof known_formats[0]; i++) { + const char *name = known_formats[i].name; + unsigned int mask = 0; + + if (xpsb_format_parse(known_formats[i].pict, &f)) { + snprintf(tag, sizeof tag, "%s rejected", name); + fail(tag); + continue; + } + snprintf(tag, sizeof tag, "%s bpp", name); + expect(tag, f.bpp, known_formats[i].bpp); + for (c = 0; c < 4; c++) { + snprintf(tag, sizeof tag, "%s bits[%u]", name, c); + expect(tag, f.bits[c], known_formats[i].bits[c]); + snprintf(tag, sizeof tag, "%s shift[%u]", name, c); + expect(tag, f.shift[c], known_formats[i].shift[c]); + if (f.bits[c]) + mask |= ((1u << f.bits[c]) - 1u) << f.shift[c]; + } + snprintf(tag, sizeof tag, "%s has_alpha", name); + expect(tag, f.has_alpha, known_formats[i].bits[0] != 0); + + printf(" %-9s bpp %2u mask %#010x\n", name, f.bpp, mask); + + for (c = 0; c < 4; c++) { + unsigned int bits = f.bits[c]; + + if (!bits) + continue; + for (v = 0; v < (1u << bits); v++) { + uint32_t raw = v << f.shift[c]; + uint32_t argb; + + argb = xpsb_pix_load(&f, (uint8_t *)&raw); + if (((argb >> (24 - 8 * c)) & 0xff) == + ref_expand(v, bits)) + continue; + snprintf(tag, sizeof tag, "%s expand ch%u", + name, c); + fail(tag); + break; + } + } + + if (f.bpp <= 16) { + for (v = 0; v < (1u << f.bpp); v++) { + if (v & ~mask) + continue; + if (roundtrip(&f, v)) + continue; + snprintf(tag, sizeof tag, "%s roundtrip %#x", + name, v); + fail(tag); + break; + } + } else { + unsigned int a, r, g, b; + + for (a = 0; a < 8; a++) + for (r = 0; r < 8; r++) + for (g = 0; g < 8; g++) + for (b = 0; b < 8; b++) { + uint32_t raw = ((probe[a] << f.shift[0]) | + (probe[r] << f.shift[1]) | + (probe[g] << f.shift[2]) | + (probe[b] << f.shift[3])) & mask; + + if (roundtrip(&f, raw)) + continue; + snprintf(tag, sizeof tag, "%s roundtrip %#x", + name, raw); + fail(tag); + a = r = g = b = 8; + } + } + } + + puts("bit expansion, r5g6b5 red:"); + xpsb_format_parse(PICT_r5g6b5, &f); + { + uint32_t raw; + + raw = 31u << 11; + expect("red 31", (xpsb_pix_load(&f, (uint8_t *)&raw) >> 16) & 0xff, + 0xff); + raw = 1u << 11; + expect("red 1", (xpsb_pix_load(&f, (uint8_t *)&raw) >> 16) & 0xff, + 0x0f); + raw = 2u << 11; + expect("red 2", (xpsb_pix_load(&f, (uint8_t *)&raw) >> 16) & 0xff, + 0x10); + raw = 63u << 5; + expect("green 63", (xpsb_pix_load(&f, (uint8_t *)&raw) >> 8) & 0xff, + 0xff); + raw = 1u << 5; + expect("green 1", (xpsb_pix_load(&f, (uint8_t *)&raw) >> 8) & 0xff, + 0x07); + raw = 0; + expect("no alpha channel", xpsb_pix_load(&f, (uint8_t *)&raw), + 0xff000000); + } + + puts("alpha of formats without an alpha channel:"); + xpsb_format_parse(PICT_x8r8g8b8, &f); + { + uint32_t raw = 0x00123456; + + expect("x8r8g8b8 opaque", xpsb_pix_load(&f, (uint8_t *)&raw), + 0xff123456); + } + xpsb_format_parse(PICT_x1r5g5b5, &f); + { + uint32_t raw = 0x8000; + + expect("x1r5g5b5 a=1", xpsb_pix_load(&f, (uint8_t *)&raw) >> 24, + 0xff); + raw = 0; + expect("x1r5g5b5 a=0", xpsb_pix_load(&f, (uint8_t *)&raw) >> 24, 0); + } + xpsb_format_parse(PICT_a8, &f); + { + uint32_t raw = 0x5a; + + expect("a8 alpha only", xpsb_pix_load(&f, (uint8_t *)&raw), + 0x5a000000); + } + xpsb_format_parse(PICT_a8b8g8r8, &f); + { + uint32_t raw = 0x80112233; + + expect("a8b8g8r8 swaps rb", xpsb_pix_load(&f, (uint8_t *)&raw), + 0x80332211); + } + + puts("unsupported formats:"); + expect("c8 rejected", (uint32_t)xpsb_format_parse(0x08040000, &f), + (uint32_t)-1); + expect("a4 rejected", (uint32_t)xpsb_format_parse(0x04014000, &f), + (uint32_t)-1); + expect("type 0 rejected", (uint32_t)xpsb_format_parse(0x20008888, &f), + (uint32_t)-1); + expect("a8 with no alpha", (uint32_t)xpsb_format_parse(0x08010000, &f), + (uint32_t)-1); +} + +static uint32_t tex4x4[16]; + +static void build_tex(struct xpsb_image *img, int filter, int umode, int vmode) +{ + unsigned int x, y; + + for (y = 0; y < 4; y++) + for (x = 0; x < 4; x++) + tex4x4[y * 4 + x] = 0xff000000u | ((x * 0x33 + 3) << 16) | + ((y * 0x33 + 5) << 8) | + ((x * 4 + y) * 0x0b + 1); + + memset(img, 0, sizeof *img); + img->base = (uint8_t *)tex4x4; + img->stride = 4 * 4; + img->w = img->h = 4; + img->filter = filter; + img->umode = umode; + img->vmode = vmode; + xpsb_format_parse(PICT_a8r8g8b8, &img->fmt); +} + +static int between(uint32_t got, uint32_t a, uint32_t b) +{ + int i; + + for (i = 0; i < 4; i++) { + int shift = 8 * i; + int cg = (got >> shift) & 0xff; + int ca = (a >> shift) & 0xff, cb = (b >> shift) & 0xff; + int lo = ca < cb ? ca : cb, hi = ca < cb ? cb : ca; + + if (cg < lo || cg > hi) + return 0; + } + return 1; +} + +static void test_sample(void) +{ + struct xpsb_image img; + char tag[64]; + int i, j, filter; + + puts("image sample, 1:1 mapping returns the texel:"); + for (filter = 0; filter < 2; filter++) { + build_tex(&img, filter, XPSB_ADDR_CLAMP, XPSB_ADDR_CLAMP); + for (j = 0; j < 4; j++) + for (i = 0; i < 4; i++) { + float u = ((float)i + 0.5f) / 4.0f; + float v = ((float)j + 0.5f) / 4.0f; + + snprintf(tag, sizeof tag, "%s texel %d,%d", + filter ? "linear" : "nearest", i, j); + expect(tag, xpsb_image_sample(&img, u, v), + tex4x4[j * 4 + i]); + } + } + printf(" texel 0,0 %#010x texel 3,3 %#010x\n", tex4x4[0], tex4x4[15]); + + puts("image sample, addressing modes:"); + build_tex(&img, XPSB_FILTER_NEAREST, XPSB_ADDR_CLAMP, XPSB_ADDR_CLAMP); + expect("clamp past right", xpsb_image_sample(&img, 1.125f, 0.125f), + tex4x4[3]); + expect("clamp before left", xpsb_image_sample(&img, -0.125f, 0.125f), + tex4x4[0]); + expect("clamp past bottom", xpsb_image_sample(&img, 0.125f, 1.125f), + tex4x4[12]); + + build_tex(&img, XPSB_FILTER_NEAREST, XPSB_ADDR_REPEAT, XPSB_ADDR_REPEAT); + expect("repeat past right", xpsb_image_sample(&img, 1.125f, 0.125f), + tex4x4[0]); + expect("repeat before left", xpsb_image_sample(&img, -0.125f, 0.125f), + tex4x4[3]); + expect("repeat past bottom", xpsb_image_sample(&img, 0.125f, 1.125f), + tex4x4[0]); + + puts("image sample, linear across the right edge:"); + build_tex(&img, XPSB_FILTER_LINEAR, XPSB_ADDR_CLAMP, XPSB_ADDR_CLAMP); + expect("clamp holds the edge", xpsb_image_sample(&img, 1.0f, 0.125f), + tex4x4[3]); + + build_tex(&img, XPSB_FILTER_LINEAR, XPSB_ADDR_REPEAT, XPSB_ADDR_REPEAT); + { + uint32_t got = xpsb_image_sample(&img, 1.0f, 0.125f); + + printf(" repeat seam %#010x between %#010x and %#010x\n", + got, tex4x4[3], tex4x4[0]); + if (!between(got, tex4x4[3], tex4x4[0])) + fail("repeat seam not a blend of both texels"); + if (got == tex4x4[3] || got == tex4x4[0]) + fail("repeat seam did not wrap"); + } +} + +enum { F_ZERO, F_ONE, F_SA, F_DA, F_1MSA, F_1MDA, F_SAT }; + +/* Porter-Duff factor pairs as the Render protocol defines them. */ +static const struct { + const char *name; + int s, d; +} ref_ops[14] = { + { "Clear", F_ZERO, F_ZERO }, + { "Src", F_ONE, F_ZERO }, + { "Dst", F_ZERO, F_ONE }, + { "Over", F_ONE, F_1MSA }, + { "OverReverse", F_1MDA, F_ONE }, + { "In", F_DA, F_ZERO }, + { "InReverse", F_ZERO, F_SA }, + { "Out", F_1MDA, F_ZERO }, + { "OutReverse", F_ZERO, F_1MSA }, + { "Atop", F_DA, F_1MSA }, + { "AtopReverse", F_1MDA, F_SA }, + { "Xor", F_1MDA, F_1MSA }, + { "Add", F_ONE, F_ONE }, + { "Saturate", F_SAT, F_ONE }, +}; + +static int ref_factor(int which, int sa, int da) +{ + int v; + + switch (which) { + case F_ONE: return 255; + case F_SA: return sa; + case F_DA: return da; + case F_1MSA: return 255 - sa; + case F_1MDA: return 255 - da; + case F_SAT: + v = sa ? (255 - da) * 255 / sa : 255; + return v > 255 ? 255 : v; + } + return 0; +} + +static unsigned int ref_mul(unsigned int a, unsigned int b) +{ + return (a * b + 127u) / 255u; +} + +static uint32_t ref_modulate(uint32_t src, unsigned int ma) +{ + uint32_t out = 0; + int i; + + for (i = 0; i < 4; i++) + out |= ref_mul((src >> (8 * i)) & 0xff, ma) << (8 * i); + return out; +} + +static uint32_t ref_blend(int op, uint32_t src, uint32_t dst) +{ + int sa = (int)(src >> 24), da = (int)(dst >> 24); + int fs = ref_factor(ref_ops[op].s, sa, da); + int fd = ref_factor(ref_ops[op].d, sa, da); + uint32_t out = 0; + int i; + + for (i = 0; i < 4; i++) { + unsigned int v = ref_mul((src >> (8 * i)) & 0xff, fs) + + ref_mul((dst >> (8 * i)) & 0xff, fd); + + out |= (v > 255 ? 255 : v) << (8 * i); + } + return out; +} + +static uint32_t comp_src[4], comp_dst[4]; + +static void comp_setup(struct xpsb_comp_state *st, const uint32_t *src) +{ + memset(st, 0, sizeof *st); + memcpy(comp_src, src, sizeof comp_src); + + xpsb_format_parse(PICT_a8r8g8b8, &st->dst.fmt); + st->dst.base = (uint8_t *)comp_dst; + st->dst.stride = 2 * 4; + st->dst.w = st->dst.h = 2; + + st->src = st->dst; + st->src.base = (uint8_t *)comp_src; + st->src.filter = XPSB_FILTER_NEAREST; + st->src.umode = st->src.vmode = XPSB_ADDR_CLAMP; + + st->scalar_mask = 1; + st->scalar = 0xff000000; +} + +static void comp_run(struct xpsb_comp_state *st) +{ + struct xpsb_quad q = { 0, 0, 2, 2, 0.0f, 0.0f, 1.0f, 1.0f, + 0.0f, 0.0f, 1.0f, 1.0f }; + + xpsb_comp_quad(st, &q); +} + +static void comp_all_ops(const char *what, const uint32_t *src, + const uint32_t *dst0) +{ + struct xpsb_comp_state st; + char tag[64]; + int op, i; + + for (op = 0; op < 14; op++) { + comp_setup(&st, src); + memcpy(comp_dst, dst0, sizeof comp_dst); + st.op = op; + comp_run(&st); + for (i = 0; i < 4; i++) { + snprintf(tag, sizeof tag, "%s %s px%d", what, + ref_ops[op].name, i); + expect(tag, comp_dst[i], ref_blend(op, src[i], dst0[i])); + } + } + printf(" %-18s Over %#010x Add %#010x Saturate %#010x\n", what, + ref_blend(3, src[0], dst0[0]), ref_blend(12, src[0], dst0[0]), + ref_blend(13, src[0], dst0[0])); +} + +static void test_composite(void) +{ + static const uint32_t opaque_src[4] = { + 0xff804020, 0xff20ff40, 0xff0080c0, 0xffffffff + }; + static const uint32_t half_src[4] = { + 0x80402010, 0x80104020, 0x80000080, 0x80808080 + }; + static const uint32_t dst0[4] = { + 0xff102030, 0xff405060, 0xff8090a0, 0xff000000 + }; + struct xpsb_comp_state st; + char tag[64]; + int op, i; + + puts("comp_quad, all 14 operators:"); + comp_all_ops("opaque src", opaque_src, dst0); + comp_all_ops("half-alpha src", half_src, dst0); + + puts("comp_quad, scalar source:"); + for (op = 0; op < 14; op++) { + comp_setup(&st, opaque_src); + memcpy(comp_dst, dst0, sizeof comp_dst); + st.op = op; + st.scalar_src = 1; + st.scalar = 0xff112233; + comp_run(&st); + for (i = 0; i < 4; i++) { + snprintf(tag, sizeof tag, "scalar src %s px%d", + ref_ops[op].name, i); + expect(tag, comp_dst[i], + ref_blend(op, 0xff112233, dst0[i])); + } + } + printf(" scalar %#010x over %#010x -> %#010x\n", 0xff112233u, dst0[0], + ref_blend(3, 0xff112233, dst0[0])); + + puts("comp_quad, scalar mask 0x80000000 modulates the source:"); + for (op = 0; op < 14; op++) { + comp_setup(&st, opaque_src); + memcpy(comp_dst, dst0, sizeof comp_dst); + st.op = op; + st.scalar = 0x80000000; + comp_run(&st); + for (i = 0; i < 4; i++) { + uint32_t src = ref_modulate(opaque_src[i], 0x80); + + snprintf(tag, sizeof tag, "masked %s px%d", + ref_ops[op].name, i); + expect(tag, comp_dst[i], ref_blend(op, src, dst0[i])); + } + } + printf(" masked src %#010x -> %#010x\n", opaque_src[0], + ref_modulate(opaque_src[0], 0x80)); +} + +/* A destination without an alpha channel must present alpha 255 to the + * blender, so the factors that read the destination alpha behave as if the + * surface were opaque. */ +static void test_composite_565(void) +{ + static const uint16_t dst0[4] = { 0x5285, 0x0000, 0xffff, 0x1234 }; + uint16_t dst[4]; + struct xpsb_comp_state st; + char tag[64]; + int op, i; + + puts("comp_quad, r5g6b5 destination reads alpha 255:"); + for (op = 0; op < 14; op++) { + static const uint32_t src[4] = { + 0xff804020, 0x80402010, 0xffffffff, 0x00000000 + }; + + memset(&st, 0, sizeof st); + memcpy(dst, dst0, sizeof dst); + memcpy(comp_src, src, sizeof comp_src); + + xpsb_format_parse(PICT_r5g6b5, &st.dst.fmt); + st.dst.base = (uint8_t *)dst; + st.dst.stride = 2 * 2; + st.dst.w = st.dst.h = 2; + + xpsb_format_parse(PICT_a8r8g8b8, &st.src.fmt); + st.src.base = (uint8_t *)comp_src; + st.src.stride = 2 * 4; + st.src.w = st.src.h = 2; + st.src.filter = XPSB_FILTER_NEAREST; + st.src.umode = st.src.vmode = XPSB_ADDR_CLAMP; + + st.scalar_mask = 1; + st.scalar = 0xff000000; + st.op = op; + comp_run(&st); + + for (i = 0; i < 4; i++) { + uint32_t d = ref_unpack565(dst0[i]); + + snprintf(tag, sizeof tag, "565 %s px%d", + ref_ops[op].name, i); + expect(tag, dst[i], + ref_pack565(ref_blend(op, src[i], d))); + } + } + + /* OverReverse with an opaque destination must leave it untouched. */ + memcpy(dst, dst0, sizeof dst); + memset(&st, 0, sizeof st); + xpsb_format_parse(PICT_r5g6b5, &st.dst.fmt); + st.dst.base = (uint8_t *)dst; + st.dst.stride = 2 * 2; + st.dst.w = st.dst.h = 2; + st.scalar_src = 1; + st.scalar_mask = 1; + st.scalar = 0xff804020; + st.op = 4; + comp_run(&st); + for (i = 0; i < 4; i++) { + snprintf(tag, sizeof tag, "565 OverReverse px%d", i); + expect(tag, dst[i], dst0[i]); + } + + memcpy(dst, dst0, sizeof dst); + st.op = 5; + comp_run(&st); + for (i = 0; i < 4; i++) { + snprintf(tag, sizeof tag, "565 In px%d", i); + expect(tag, dst[i], ref_pack565(0xff804020)); + } + printf(" In stored %#06x for source %#010x\n", dst[0], 0xff804020u); +} + +static void conv_rgb(const struct xpsb_yuv_conv *c, int y, int u, int v, + int *rgb) +{ + int i; + + for (i = 0; i < 3; i++) { + int32_t t = c->c[i][0] * y + c->c[i][1] * u + + c->c[i][2] * v + c->k[i]; + + t >>= c->shift[i]; + rgb[i] = t < 0 ? 0 : (t > 255 ? 255 : (int)t); + } +} + +static void check_black_white(const char *what, const struct xpsb_yuv_conv *c) +{ + static const char *ch[3] = { "r", "g", "b" }; + char tag[64]; + int rgb[3], i; + + conv_rgb(c, 16, 128, 128, rgb); + printf(" %-8s Y16 -> %3d %3d %3d\n", what, rgb[0], rgb[1], rgb[2]); + for (i = 0; i < 3; i++) { + snprintf(tag, sizeof tag, "%s black %s", what, ch[i]); + expect_near(tag, rgb[i], 0, 2); + } + + conv_rgb(c, 235, 128, 128, rgb); + printf(" %-8s Y235 -> %3d %3d %3d\n", what, rgb[0], rgb[1], rgb[2]); + for (i = 0; i < 3; i++) { + snprintf(tag, sizeof tag, "%s white %s", what, ch[i]); + expect_near(tag, rgb[i], 255, 2); + } +} + +static const float bt_601[11] = { + 1.0f, 0.0f, 1.4075f, + 1.0f, -0.3455f, -0.7169f, + 1.0f, 1.7790f, 0.0f, + -16.0f, 0.0f +}; + +static void bt601_scaled(float *out) +{ + int i; + + for (i = 0; i < 9; i++) + out[i] = bt_601[i] / 219.0f; + out[9] = bt_601[9]; + out[10] = bt_601[10]; +} + +/* Fixed-point BT.601 in the eight-bit coefficient range the SGX words carry; + * blue needs a smaller shift because its U term exceeds 127 otherwise. */ +static const struct { + int cy, cu, cv, k, shift; +} planar_coeffs[3] = { + { 75, 0, 102, -14266, 6 }, + { 75, -25, -52, 8671, 6 }, + { 37, 65, 0, -8862, 5 }, +}; + +static void test_yuv_conv(void) +{ + struct xpsb_yuv_conv conv, def; + uint32_t words[9]; + float scaled[11]; + char tag[64]; + int i; + + puts("yuv_conv_packed, BT.601 limited range:"); + bt601_scaled(scaled); + xpsb_yuv_conv_packed(scaled, &conv); + printf(" row0 %d %d %d k %d >>%d\n", conv.c[0][0], conv.c[0][1], + conv.c[0][2], conv.k[0], conv.shift[0]); + check_black_white("packed", &conv); + + puts("yuv_conv_default matches the same matrix:"); + xpsb_yuv_conv_default(&def); + if (memcmp(&def, &conv, sizeof def)) + fail("default differs from the explicit BT.601 matrix"); + check_black_white("default", &def); + + puts("yuv_conv_planar, packed SGX words:"); + for (i = 0; i < 3; i++) { + words[3 * i + 0] = ((uint32_t)(planar_coeffs[i].cy & 0xff) << 24) | + (uint32_t)(planar_coeffs[i].cu & 0xff); + words[3 * i + 1] = (uint32_t)(planar_coeffs[i].cv & 0xff) << 8; + words[3 * i + 2] = ((uint32_t)(planar_coeffs[i].k & 0xffff) << 4) | + (uint32_t)(planar_coeffs[i].shift & 0xf); + } + xpsb_yuv_conv_planar(words, &conv); + for (i = 0; i < 3; i++) { + printf(" row%d %d %d %d k %d >>%d\n", i, conv.c[i][0], + conv.c[i][1], conv.c[i][2], conv.k[i], conv.shift[i]); + snprintf(tag, sizeof tag, "row%d cY", i); + expect(tag, (uint32_t)conv.c[i][0], (uint32_t)planar_coeffs[i].cy); + snprintf(tag, sizeof tag, "row%d cU", i); + expect(tag, (uint32_t)conv.c[i][1], (uint32_t)planar_coeffs[i].cu); + snprintf(tag, sizeof tag, "row%d cV", i); + expect(tag, (uint32_t)conv.c[i][2], (uint32_t)planar_coeffs[i].cv); + snprintf(tag, sizeof tag, "row%d const", i); + expect(tag, (uint32_t)conv.k[i], (uint32_t)planar_coeffs[i].k); + snprintf(tag, sizeof tag, "row%d shift", i); + expect(tag, (uint32_t)conv.shift[i], + (uint32_t)planar_coeffs[i].shift); + } + check_black_white("planar", &conv); + + puts("yuv_planes:"); + expect("YUY2", (uint32_t)xpsb_yuv_planes(XPSB_FOURCC_YUY2), 1); + expect("UYVY", (uint32_t)xpsb_yuv_planes(XPSB_FOURCC_UYVY), 1); + expect("NV12", (uint32_t)xpsb_yuv_planes(XPSB_FOURCC_NV12), 2); + expect("YV12", (uint32_t)xpsb_yuv_planes(XPSB_FOURCC_YV12), 3); + expect("I420", (uint32_t)xpsb_yuv_planes(XPSB_FOURCC_I420), 3); + expect("junk", (uint32_t)xpsb_yuv_planes(0x12345678), 0); +} + +#define VW 4 +#define VH 4 +#define Y_HI 235 +#define Y_LO 16 + +static uint8_t yuy2_buf[VH * VW * 2]; +static uint8_t uyvy_buf[VH * VW * 2]; +static uint8_t yplane[VH * VW]; +static uint8_t uplane[(VH / 2) * (VW / 2)]; +static uint8_t vplane[(VH / 2) * (VW / 2)]; +static uint8_t nv12_uv[(VH / 2) * VW]; + +static void build_yuv_sources(void) +{ + int x, y; + + for (y = 0; y < VH; y++) + for (x = 0; x < VW; x += 2) { + uint8_t *p = yuy2_buf + y * VW * 2 + x * 2; + + p[0] = Y_HI; p[1] = 128; p[2] = Y_LO; p[3] = 128; + p = uyvy_buf + y * VW * 2 + x * 2; + p[0] = 128; p[1] = Y_HI; p[2] = 128; p[3] = Y_LO; + } + + for (y = 0; y < VH; y++) + for (x = 0; x < VW; x++) + yplane[y * VW + x] = (x & 1) ? Y_LO : Y_HI; + + memset(uplane, 128, sizeof uplane); + memset(vplane, 128, sizeof vplane); + memset(nv12_uv, 128, sizeof nv12_uv); +} + +static void expect_rgb(const char *what, uint32_t argb, int r, int g, int b) +{ + char tag[80]; + + snprintf(tag, sizeof tag, "%s r", what); + expect_near(tag, (int)((argb >> 16) & 0xff), r, 2); + snprintf(tag, sizeof tag, "%s g", what); + expect_near(tag, (int)((argb >> 8) & 0xff), g, 2); + snprintf(tag, sizeof tag, "%s b", what); + expect_near(tag, (int)(argb & 0xff), b, 2); +} + +static void blit_case(const char *what, const struct xpsb_yuv_src *src, + const struct xpsb_yuv_conv *conv) +{ + uint32_t dst[VH * VW]; + struct xpsb_image img; + char tag[80]; + int x, y; + + memset(dst, 0, sizeof dst); + memset(&img, 0, sizeof img); + img.base = (uint8_t *)dst; + img.stride = VW * 4; + img.w = VW; + img.h = VH; + xpsb_format_parse(PICT_x8r8g8b8, &img.fmt); + + xpsb_yuv_blit(&img, 0, 0, VW, VH, 0.0f, 0.0f, 1.0f, 1.0f, src, conv); + + for (y = 0; y < VH; y++) + for (x = 0; x < VW; x++) { + int want = (x & 1) ? 0 : 255; + + snprintf(tag, sizeof tag, "%s %d,%d", what, x, y); + expect_rgb(tag, dst[y * VW + x], want, want, want); + } + printf(" %-5s row0 %06x %06x %06x %06x\n", what, + dst[0] & 0xffffff, dst[1] & 0xffffff, dst[2] & 0xffffff, + dst[3] & 0xffffff); +} + +static void test_yuv_blit(void) +{ + struct xpsb_yuv_conv conv; + struct xpsb_yuv_src src; + float scaled[11]; + + bt601_scaled(scaled); + xpsb_yuv_conv_packed(scaled, &conv); + build_yuv_sources(); + + puts("yuv_blit 1:1, even columns white and odd columns black:"); + + memset(&src, 0, sizeof src); + src.w = VW; + src.h = VH; + src.filter = XPSB_FILTER_NEAREST; + src.plane[0] = yuy2_buf; + src.stride[0] = VW * 2; + src.fourcc = XPSB_FOURCC_YUY2; + blit_case("YUY2", &src, &conv); + + src.plane[0] = uyvy_buf; + src.fourcc = XPSB_FOURCC_UYVY; + blit_case("UYVY", &src, &conv); + + src.plane[0] = yplane; + src.stride[0] = VW; + src.plane[1] = uplane; + src.stride[1] = VW / 2; + src.plane[2] = vplane; + src.stride[2] = VW / 2; + src.fourcc = XPSB_FOURCC_YV12; + blit_case("YV12", &src, &conv); + + src.plane[1] = nv12_uv; + src.stride[1] = VW; + src.plane[2] = NULL; + src.fourcc = XPSB_FOURCC_NV12; + blit_case("NV12", &src, &conv); + + puts("yuv_blit 2x upscale:"); + { + uint32_t dst[8 * 8 + 16]; + struct xpsb_image img; + char tag[64]; + int i, x, y; + + for (i = 0; i < (int)(sizeof dst / sizeof dst[0]); i++) + dst[i] = 0xdeadbeef; + + memset(&img, 0, sizeof img); + img.base = (uint8_t *)dst; + img.stride = 8 * 4; + img.w = img.h = 8; + xpsb_format_parse(PICT_x8r8g8b8, &img.fmt); + + memset(&src, 0, sizeof src); + src.w = VW; + src.h = VH; + src.filter = XPSB_FILTER_NEAREST; + src.plane[0] = yuy2_buf; + src.stride[0] = VW * 2; + src.fourcc = XPSB_FOURCC_YUY2; + + xpsb_yuv_blit(&img, 0, 0, 8, 8, 0.0f, 0.0f, 1.0f, 1.0f, &src, + &conv); + + for (y = 0; y < 8; y++) + for (x = 0; x < 8; x++) { + int want = ((x >> 1) & 1) ? 0 : 255; + + snprintf(tag, sizeof tag, "2x %d,%d", x, y); + expect_rgb(tag, dst[y * 8 + x], want, want, want); + } + expect_rgb("2x corner 0,0", dst[0], 255, 255, 255); + expect_rgb("2x corner 7,7", dst[63], 0, 0, 0); + for (i = 64; i < (int)(sizeof dst / sizeof dst[0]); i++) + if (dst[i] != 0xdeadbeef) + fail("2x upscale wrote past the destination"); + printf(" row0 %06x %06x %06x %06x %06x %06x %06x %06x\n", + dst[0] & 0xffffff, dst[1] & 0xffffff, dst[2] & 0xffffff, + dst[3] & 0xffffff, dst[4] & 0xffffff, dst[5] & 0xffffff, + dst[6] & 0xffffff, dst[7] & 0xffffff); + } + + puts("yuv_blit clipped at a negative offset:"); + { + uint32_t dst[VH * VW]; + struct xpsb_image img; + char tag[64]; + int x, y; + + for (x = 0; x < VH * VW; x++) + dst[x] = 0xcafebabe; + + memset(&img, 0, sizeof img); + img.base = (uint8_t *)dst; + img.stride = VW * 4; + img.w = VW; + img.h = VH; + xpsb_format_parse(PICT_x8r8g8b8, &img.fmt); + + memset(&src, 0, sizeof src); + src.w = VW; + src.h = VH; + src.filter = XPSB_FILTER_NEAREST; + src.plane[0] = yuy2_buf; + src.stride[0] = VW * 2; + src.fourcc = XPSB_FOURCC_YUY2; + + xpsb_yuv_blit(&img, -2, -2, VW, VH, 0.0f, 0.0f, 1.0f, 1.0f, + &src, &conv); + + for (y = 0; y < VH; y++) + for (x = 0; x < VW; x++) { + snprintf(tag, sizeof tag, "clipped %d,%d", x, y); + if (x < 2 && y < 2) + expect_rgb(tag, dst[y * VW + x], + (x & 1) ? 0 : 255, + (x & 1) ? 0 : 255, + (x & 1) ? 0 : 255); + else + expect(tag, dst[y * VW + x], 0xcafebabe); + } + printf(" wrote %06x %06x, kept %#010x\n", dst[0] & 0xffffff, + dst[1] & 0xffffff, dst[15]); + } +} + +int main(void) +{ + test_format(); + test_sample(); + test_composite(); + test_composite_565(); + test_yuv_conv(); + test_yuv_blit(); + + printf("\n%s: %d failure%s\n", failures ? "FAILED" : "ok", failures, + failures == 1 ? "" : "s"); + return failures != 0; +} diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/test_vidshader.c mesa-26.2.2/src/gallium/drivers/sgx/test_vidshader.c --- mesa-26.2.2.orig/src/gallium/drivers/sgx/test_vidshader.c 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/test_vidshader.c 2026-09-08 10:57:36.684595162 +0200 @@ -0,0 +1,271 @@ +/* Self-test for the GPU video shader blob and the two coefficient routes. + * + * Nothing here touches hardware. The expected shader words are restated from + * the .rodata dump in disasm/video-stream.md, the expected coefficient + * registers from the same document, and the coefficient values are + * cross-checked against the CPU fallback in xpsb_yuv.c so both paths provably + * see the same conversion. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#include +#include + +#include "xpsb_vidshader.h" +#include "xpsb_yuv.h" + +static int failures; + +static void expect(const char *what, uint32_t got, uint32_t want) +{ + if (got != want) { + printf(" FAIL %-28s got %#010x want %#010x\n", what, got, + want); + failures++; + } +} + +static void expect_ptr(const char *what, const void *got, const void *want) +{ + if (got != want) { + printf(" FAIL %-28s got %p want %p\n", what, got, want); + failures++; + } +} + +static void test_blob(void) +{ + printf("USSE static buffer\n"); + + expect("blob bytes", XPSB_USSE_BLOB_BYTES, 440); + expect("blob dwords", XPSB_USSE_BLOB_DWORDS, 110); + expect("packed offset", XPSB_USSE_OFF_PACKED, 0x110); + expect("nv12 offset", XPSB_USSE_OFF_NV12, 0x180); + expect("planar offset", XPSB_USSE_OFF_PLANAR, 0x1a0); + expect("lengths sum", XPSB_USSE_OFF_PLANAR + XPSB_USSE_LEN_PLANAR, + XPSB_USSE_BLOB_BYTES); + expect("packed length", XPSB_USSE_LEN_PACKED, + XPSB_USSE_OFF_NV12 - XPSB_USSE_OFF_PACKED); + expect("nv12 length", XPSB_USSE_LEN_NV12, + XPSB_USSE_OFF_PLANAR - XPSB_USSE_OFF_NV12); + + /* First and last instruction word of each program, .rodata 0xd1f0. */ + expect("packed[0].lo", xpsb_usse_blob[0x110 / 4], 0xa0400002); + expect("packed[0].hi", xpsb_usse_blob[0x110 / 4 + 1], 0x400015bc); + expect("packed end.lo", xpsb_usse_blob[0x178 / 4], 0x00000000); + expect("packed end.hi", xpsb_usse_blob[0x178 / 4 + 1], 0xf8000140); + expect("nv12[0].lo", xpsb_usse_blob[0x180 / 4], 0xa0000081); + expect("nv12[0].hi", xpsb_usse_blob[0x180 / 4 + 1], 0xb8a04841); + expect("nv12 pad.hi", xpsb_usse_blob[0x198 / 4 + 1], 0xf8000140); + expect("planar[0].lo", xpsb_usse_blob[0x1a0 / 4], 0xa0000101); + expect("planar[2].hi", xpsb_usse_blob[0x1b0 / 4 + 1], 0xb8844851); + + /* NV12 and YV12 differ only in the src1 field of the FIRH words. */ + expect("planar src1 delta", + xpsb_usse_blob[0x1a0 / 4] - xpsb_usse_blob[0x180 / 4], 0x80); +} + +static void one_select(const char *what, uint32_t fourcc, uint32_t off, + unsigned dwords) +{ + unsigned n = 0; + const uint32_t *p = xpsb_vidshader(fourcc, &n); + + expect_ptr(what, p, &xpsb_usse_blob[off / 4]); + expect(what, n, dwords); + expect(what, xpsb_vidshader_offset(fourcc), off); +} + +static void test_select(void) +{ + unsigned n = 0xdeadbeef; + + printf("shader selection\n"); + + one_select("YUY2", XPSB_FOURCC_YUY2, XPSB_USSE_OFF_PACKED, + XPSB_USSE_LEN_PACKED / 4); + one_select("UYVY", XPSB_FOURCC_UYVY, XPSB_USSE_OFF_PACKED, + XPSB_USSE_LEN_PACKED / 4); + one_select("NV12", XPSB_FOURCC_NV12, XPSB_USSE_OFF_NV12, + XPSB_USSE_LEN_NV12 / 4); + one_select("YV12", XPSB_FOURCC_YV12, XPSB_USSE_OFF_PLANAR, + XPSB_USSE_LEN_PLANAR / 4); + one_select("I420", XPSB_FOURCC_I420, XPSB_USSE_OFF_PLANAR, + XPSB_USSE_LEN_PLANAR / 4); + + /* The blob has no default case; an unknown FOURCC must be rejected. */ + expect_ptr("RGB16 rejected", xpsb_vidshader(0x36314752, &n), NULL); + expect("RGB16 dwords kept", n, 0xdeadbeef); + expect_ptr("zero rejected", xpsb_vidshader(0, NULL), NULL); + expect("zero offset", xpsb_vidshader_offset(0), ~0u); +} + +/* conversionData as psbSetupConversionData builds it: BT.601 over yRange. */ +static void ref_conversion_data(float *d) +{ + static const float bt601[9] = { + 1.0f, 0.0f, 1.4075f, + 1.0f, -0.3455f, -0.7169f, + 1.0f, 1.7790f, 0.0f + }; + int i; + + for (i = 0; i < 9; i++) + d[i] = bt601[i] / 219.0f; + d[9] = -16.0f; + d[10] = 2.2f; +} + +static void test_packed_route(void) +{ + struct xpsb_vidshader_sa_pds pds; + float conv[XPSB_VIDSHADER_SA_PACKED]; + uint32_t buf[XPSB_VIDSHADER_SA_PACKED]; + struct xpsb_yuv_conv from_blob, from_ddx; + unsigned i; + + printf("packed coefficient route\n"); + + expect("dma ctrl n=11", xpsb_vidshader_sa_dma_ctrl(11), + 0x80000000u | (10u << 21) | 10u); + expect("dma ctrl n=11 literal", xpsb_vidshader_sa_dma_ctrl(11), + 0x8140000au); + expect("dma ctrl n=16", xpsb_vidshader_sa_dma_ctrl(16), + 0x80000000u | (15u << 21) | 15u); + + ref_conversion_data(conv); + memset(buf, 0xa5, sizeof buf); + expect("consts copied", xpsb_vidshader_emit_sa_consts(buf, conv, + XPSB_VIDSHADER_SA_PACKED), XPSB_VIDSHADER_SA_PACKED); + if (memcmp(buf, conv, sizeof conv) != 0) { + printf(" FAIL constant block is not verbatim\n"); + failures++; + } + + expect("emit sa pds", (uint32_t)xpsb_vidshader_emit_sa_pds(&pds, conv, + XPSB_VIDSHADER_SA_PACKED, 0x67676767u), 0); + expect("pds[0] const addr", pds.prog[0], 0x67676767u); + expect("pds[1] dma ctrl", pds.prog[1], 0x8140000au); + for (i = 2; i < 12; i++) + expect("pds data zeroed", pds.prog[i], 0); + expect("pds[12] doutd", pds.prog[12], XPSB_PDS_DOUTD); + expect("pds[13] halt", pds.prog[13], XPSB_PDS_HALT); + expect("pds n_prog", pds.n_prog, 14); + expect("pds data dwords", pds.data_dwords, 12); + expect("pds n_attrs", pds.n_attrs, XPSB_VIDSHADER_SA_PACKED); + + /* Planar formats pass n = 0: bare terminator, nothing else written. */ + expect("emit sa pds n=0", + (uint32_t)xpsb_vidshader_emit_sa_pds(&pds, NULL, 0, 0), 0); + for (i = 0; i < 8; i++) + expect("n=0 data zeroed", pds.prog[i], 0); + expect("n=0 halt", pds.prog[8], XPSB_PDS_HALT); + expect("n=0 data dwords", pds.data_dwords, 8); + expect("n=0 n_attrs", pds.n_attrs, 1); + + /* sa0..sa10 hold the DDX floats bit for bit, so the CPU fallback and + * the shader work from one and the same matrix. */ + xpsb_yuv_conv_packed((const float *)buf, &from_blob); + xpsb_yuv_conv_packed(conv, &from_ddx); + if (memcmp(&from_blob, &from_ddx, sizeof from_blob) != 0) { + printf(" FAIL sa block disagrees with conversionData\n"); + failures++; + } + printf(" sa0..sa10 verbatim, R row %d %d %d >>%d\n", + from_blob.c[0][0], from_blob.c[0][1], from_blob.c[0][2], + from_blob.shift[0]); +} + +/* psb_pack_coeffs, restated from psb_video.c:1529. */ +static void ref_pack_coeffs(uint32_t *out, const int *c) +{ + int i; + + for (i = 0; i < 3; i++) { + const int *g = &c[5 * i]; + + out[3 * i] = ((uint32_t)(g[0] & 0xff) << 24) | + ((uint32_t)(g[1] & 0xff) << 0); + out[3 * i + 1] = (uint32_t)(g[2] & 0xff) << 8; + out[3 * i + 2] = ((uint32_t)(g[3] & 0xffff) << 4) | + (uint32_t)(g[4] & 0xf); + } +} + +static void test_planar_route(void) +{ + /* Y, U, V, Const, Shift per channel, in the ranges psb_convert_coeffs + * guarantees: coefficients signed 8 bit, constant signed 16 bit, shift + * 4 bit. BT.601 halved until it fits, so the G row carries the negative + * values that exercise the sign extension both sides must agree on. */ + static const int ddx_coeffs[15] = { + 74, 0, 102, -14265, 6, + 74, -25, -52, 8680, 6, + 37, 64, 0, -8858, 5 + }; + static const uint32_t want_reg[XPSB_VIDSHADER_COEFF_PAIRS] = { + 0x0a9c, + 0x0a84, 0x0a88, 0x0a8c, + 0x0a90, 0x0a94, 0x0a98, + 0x0b20, 0x0b24, 0x0b28 + }; + uint32_t sgx[9], out[XPSB_VIDSHADER_COEFF_DWORDS]; + uint32_t stream[9]; + struct xpsb_yuv_conv from_regs, from_ddx; + unsigned i; + + printf("planar coefficient route\n"); + + ref_pack_coeffs(sgx, ddx_coeffs); + memset(out, 0x5a, sizeof out); + expect("pairs emitted", xpsb_vidshader_emit_coeff_regs(out, sgx), + XPSB_VIDSHADER_COEFF_DWORDS); + + expect("reg[0]", out[0], want_reg[0]); + expect("val[0]", out[1], 3); + for (i = 0; i < 9; i++) { + expect("reg", out[2 + 2 * i], want_reg[1 + i]); + expect("val verbatim", out[3 + 2 * i], sgx[i]); + stream[i] = out[3 + 2 * i]; + } + + /* Unpack straight out of the emitted stream and compare with what the + * CPU path makes of the same nine words. */ + xpsb_yuv_conv_planar(stream, &from_regs); + xpsb_yuv_conv_planar(sgx, &from_ddx); + if (memcmp(&from_regs, &from_ddx, sizeof from_regs) != 0) { + printf(" FAIL register stream disagrees with sgx_coeffs\n"); + failures++; + } + + for (i = 0; i < 3; i++) { + const int *g = &ddx_coeffs[5 * i]; + + expect("unpacked Y", (uint32_t)from_regs.c[i][0], + (uint32_t)g[0]); + expect("unpacked U", (uint32_t)from_regs.c[i][1], + (uint32_t)g[1]); + expect("unpacked V", (uint32_t)from_regs.c[i][2], + (uint32_t)g[2]); + expect("unpacked const", (uint32_t)from_regs.k[i], + (uint32_t)g[3]); + expect("unpacked shift", (uint32_t)from_regs.shift[i], + (uint32_t)g[4]); + } + printf(" 0x%04x=%u then R %#010x %#010x %#010x\n", out[0], out[1], + out[3], out[5], out[7]); +} + +int main(void) +{ + test_blob(); + test_select(); + test_packed_route(); + test_planar_route(); + + printf("\n%s: %d failure%s\n", failures ? "FAILED" : "ok", failures, + failures == 1 ? "" : "s"); + return failures != 0; +} diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/tgsi_parse.c mesa-26.2.2/src/gallium/drivers/sgx/tgsi_parse.c --- mesa-26.2.2.orig/src/gallium/drivers/sgx/tgsi_parse.c 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/tgsi_parse.c 2026-09-08 10:57:36.681811569 +0200 @@ -0,0 +1,378 @@ +/* TGSI token decoding - see tgsi_parse.h. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: GPL-2.0-only + */ +#include "tgsi_parse.h" + +#include +#include +#include + +/* Field extraction. Transcribed from Mesa's struct declarations, LSB first, + * which is the order those bit-fields are laid out in by every compiler Mesa + * supports. See the header for why this is not a struct. */ +#define F(w, lo, bits) (((w) >> (lo)) & ((1u << (bits)) - 1u)) + +/* struct tgsi_header / tgsi_processor */ +#define HDR_TYPE(w) F(w, 0, 4) + +/* struct tgsi_instruction */ +#define INS_TYPE(w) F(w, 0, 4) +#define INS_NRTOK(w) F(w, 4, 8) +#define INS_OPCODE(w) F(w, 12, 8) +#define INS_SAT(w) F(w, 20, 1) +#define INS_NDST(w) F(w, 21, 2) +#define INS_NSRC(w) F(w, 23, 4) +#define INS_LABEL(w) F(w, 27, 1) +#define INS_TEXTURE(w) F(w, 28, 1) + +/* struct tgsi_instruction_texture */ +#define TEX_TARGET(w) F(w, 0, 8) +#define TEX_NOFFSETS(w) F(w, 8, 4) + +/* struct tgsi_dst_register */ +static int tgsi_probe(void) +{ + static int on = -1; + + if (on < 0) + on = getenv("SGX_TGSI_PROBE") != NULL; + return on; +} + +#define DST_FILE(w) F(w, 0, 4) +#define DST_MASK(w) F(w, 4, 4) +#define DST_INDIRECT(w) F(w, 8, 1) +#define DST_DIM(w) F(w, 9, 1) +#define DST_INDEX(w) F(w, 10, 16) + +/* struct tgsi_src_register */ +#define SRC_FILE(w) F(w, 0, 4) +#define SRC_INDIRECT(w) F(w, 4, 1) +#define SRC_DIM(w) F(w, 5, 1) +#define SRC_INDEX(w) F(w, 6, 16) +#define SRC_SWZ_X(w) F(w, 22, 2) +#define SRC_SWZ_Y(w) F(w, 24, 2) +#define SRC_SWZ_Z(w) F(w, 26, 2) +#define SRC_SWZ_W(w) F(w, 28, 2) +#define SRC_ABS(w) F(w, 30, 1) +#define SRC_NEG(w) F(w, 31, 1) + +/* struct tgsi_dimension - the second index of a two-dimensional register, + * which is what nir_to_tgsi gives every uniform: CONST[buffer][register]. */ +#define DIM_INDIRECT(w) F(w, 0, 1) +#define DIM_DIM(w) F(w, 1, 1) +#define DIM_INDEX(w) F(w, 16, 16) + +/* struct tgsi_immediate */ +#define IMM_TYPE(w) F(w, 0, 4) +#define IMM_NRTOK(w) F(w, 4, 14) + +/* struct tgsi_declaration. The optional tokens follow the range in the order + * Mesa's own parser reads them: dimension, interpolation, semantic. */ +#define DCL_FILE(w) F(w, 12, 4) +#define DCL_DIM(w) F(w, 20, 1) +#define DCL_SEM(w) F(w, 21, 1) +#define DCL_INTERP(w) F(w, 22, 1) +#define DCL_FIRST(w) F(w, 0, 16) +#define DCL_LAST(w) F(w, 16, 16) +#define SEM_NAME(w) F(w, 0, 8) +#define SEM_INDEX(w) F(w, 8, 16) +#define TGSI_FILE_IN 2 +#define TGSI_FILE_OUT 3 +#define TGSI_FILE_SV 8 + +#define TOKEN_DECLARATION 0 +#define TOKEN_IMMEDIATE 1 +#define TOKEN_INSTRUCTION 2 +#define TOKEN_PROPERTY 3 + +/* Which of the several checks behind TGSI_PARSE_BAD_TOKEN actually fired. + * "unhandled token type" was the answer for indirect addressing, a second + * destination and an unknown token alike, which are different features with + * different work behind them. Set on the way out; the compile lock in the + * driver is what makes one static safe. */ +static const char *tgsi_bad_reason = "unhandled token type"; + +#define BAD(why) (tgsi_bad_reason = (why), TGSI_PARSE_BAD_TOKEN) + +const char *tgsi_parse_status_name(enum tgsi_parse_status st) +{ + switch (st) { + case TGSI_PARSE_OK: return "ok"; + case TGSI_PARSE_TRUNCATED: return "token stream is truncated"; + case TGSI_PARSE_BAD_TOKEN: return tgsi_bad_reason; + case TGSI_PARSE_TOO_MANY: return "too many instructions"; + case TGSI_PARSE_TOO_MANY_IMM: return "too many immediates"; + case TGSI_PARSE_BAD_IMMEDIATE: return "immediate is not four floats"; + case TGSI_PARSE_BAD_HEADER: return "bad header"; + } + return "?"; +} + +/* An index is 16 bits and declared signed in Mesa's struct, so it has to be + * sign-extended. Reading it as unsigned makes a relative index into a very + * large positive one, which indexes out of the register file rather than + * failing. */ +static int index_of(uint32_t raw) +{ + int v = (int)raw; + + return (v & 0x8000) ? (v | ~0xffff) : v; +} + +enum tgsi_parse_status tgsi_parse_tokens(const uint32_t *tokens, unsigned ntok, + struct tgsi_insn *out, unsigned cap, + unsigned *ninsns, int *stage) +{ + return tgsi_parse_tokens_io(tokens, ntok, out, cap, ninsns, stage, + NULL); +} + +/* Record one declaration's semantic against every register it covers. */ +static void io_note(struct tgsi_io *io, unsigned file, unsigned first, + unsigned last, unsigned name, unsigned index) +{ + uint8_t *sem, *idx; + unsigned *n, r; + + if (!io) + return; + if (file == TGSI_FILE_IN) { + sem = io->in_semantic; idx = io->in_index; n = &io->nin; + } else if (file == TGSI_FILE_OUT) { + sem = io->out_semantic; idx = io->out_index; n = &io->nout; + } else { + return; + } + for (r = first; r <= last && r < TGSI_PARSE_MAX_IO; r++) { + sem[r] = (uint8_t)name; + idx[r] = (uint8_t)(index + (r - first)); + if (r + 1 > *n) + *n = r + 1; + } +} + +/* A system value's declaration index is only meaningful against its + * declaration, so the operand is renamed to the semantic the declaration + * carries (tgsi_to_uir.h: TGSI_SEM_FACE), and the translator never sees the + * index at all. Declarations precede instructions in a token stream. */ +static void sv_note(uint8_t *sv_sem, unsigned first, unsigned last, + unsigned name) +{ + unsigned r; + + for (r = first; r <= last && r < TGSI_PARSE_MAX_IO; r++) + sv_sem[r] = (uint8_t)name; +} + +enum tgsi_parse_status tgsi_parse_tokens_io(const uint32_t *tokens, + unsigned ntok, + struct tgsi_insn *out, unsigned cap, + unsigned *ninsns, int *stage, + struct tgsi_io *io) +{ + float imm[TGSI_PARSE_MAX_IMM][4]; + uint8_t sv_sem[TGSI_PARSE_MAX_IO]; + unsigned nimm = 0, i = 0, n = 0; + + memset(sv_sem, 0xff, sizeof sv_sem); + if (io) { + memset(io, 0, sizeof *io); + memset(io->in_semantic, 0xff, sizeof io->in_semantic); + memset(io->out_semantic, 0xff, sizeof io->out_semantic); + } + + if (!tokens || !out || !ninsns || ntok < 2) + return TGSI_PARSE_BAD_HEADER; + *ninsns = 0; + + /* The header is two tokens: a tgsi_header and a tgsi_processor. The + * processor is what says vertex or fragment. */ + if (stage) + *stage = HDR_TYPE(tokens[1]) == 0 ? 0 : 1; + i = 2; + + while (i < ntok) { + uint32_t t = tokens[i]; + unsigned type = INS_TYPE(t); + + if (type == TOKEN_INSTRUCTION) { + unsigned nrtok = INS_NRTOK(t); + unsigned ndst = INS_NDST(t), nsrc = INS_NSRC(t); + unsigned k, at = i + 1; + struct tgsi_insn *o; + + /* An instruction's NrTokens counts the tokens that + * follow it, not itself: tgsi_default_instruction() + * starts it at zero and instruction_grow() raises it + * once per token added. So END, which has none, has a + * count of zero and is not truncated. A declaration + * and an immediate are the other way round - Mesa's + * own parser reads their data as NrTokens - 1 - which + * is why they are stepped over differently below. */ + if (i + 1 + nrtok > ntok) + return TGSI_PARSE_TRUNCATED; + if (n >= cap) + return TGSI_PARSE_TOO_MANY; + unsigned char target = 0; + + /* A label or texture token sits between the + * instruction and its registers. Skipping it would + * decode the label as a destination register. The + * texture token's target is kept: a volume or cube + * sample reads three coordinate components. */ + if (INS_LABEL(t)) + at++; + /* The texture token names the target, which decides + * whether the sample is one this part takes and how + * many coordinate components it reads. An offset + * list (textureOffset) follows it, one token each, + * and has no hardware form here. */ + if (INS_TEXTURE(t)) { + if (at >= i + 1 + nrtok) + return TGSI_PARSE_TRUNCATED; + target = (unsigned char)TEX_TARGET(tokens[at]); + if (TEX_NOFFSETS(tokens[at])) + return BAD("a texture offset"); + at++; + } + if (ndst > 1 || nsrc > 4) + return BAD("more than one destination or four sources"); + /* Each source may carry a dimension token of its own, + * so this is the minimum, not the exact size; the + * reads below are bounds-checked one at a time. */ + if (at + ndst + nsrc > i + 1 + nrtok) + return TGSI_PARSE_TRUNCATED; + + o = &out[n]; + memset(o, 0, sizeof(*o)); + o->opcode = INS_OPCODE(t); + o->saturate = (unsigned char)INS_SAT(t); + o->tex_target = target; + if (ndst) { + uint32_t d = tokens[at++]; + + o->dst.file = (enum tgsi_file)DST_FILE(d); + o->dst.index = (unsigned)index_of(DST_INDEX(d)); + o->dst.writemask = (unsigned char)DST_MASK(d); + if (DST_INDIRECT(d) || DST_DIM(d)) + return BAD("indirect or dimensioned destination"); + } + for (k = 0; k < nsrc; k++) { + uint32_t sw; + struct tgsi_src *s = &o->src[k]; + + if (at >= i + 1 + nrtok) + return TGSI_PARSE_TRUNCATED; + sw = tokens[at++]; + if (SRC_INDIRECT(sw)) + return BAD("indirectly addressed source"); + /* A dimension names the constant buffer. One + * buffer is bound here, so buffer 0 is the + * whole of what can be decoded - and it is + * what every uniform in a nir_to_tgsi program + * is written as, so refusing it outright + * refused every shader with a uniform. */ + if (SRC_DIM(sw)) { + uint32_t dim; + + if (at >= i + 1 + nrtok) + return TGSI_PARSE_TRUNCATED; + dim = tokens[at++]; + if (DIM_INDIRECT(dim) || DIM_DIM(dim) || + index_of(DIM_INDEX(dim)) != 0) + return BAD("a constant buffer other than 0"); + } + s->file = (enum tgsi_file)SRC_FILE(sw); + s->index = (unsigned)index_of(SRC_INDEX(sw)); + s->swizzle = (unsigned char) + (SRC_SWZ_X(sw) | (SRC_SWZ_Y(sw) << 2) | + (SRC_SWZ_Z(sw) << 4) | + (SRC_SWZ_W(sw) << 6)); + /* SGX_TGSI_PROBE prints the token a source + * decodes from. The field offsets are a + * reverse engineered struct layout, so being + * able to check one against its decode is + * worth the two lines. */ + if (tgsi_probe()) + fprintf(stderr, "tgsi: src token %08x -> " + "file %u index %u swz %02x\n", + sw, (unsigned)s->file, + s->index, + (unsigned)s->swizzle); + s->absolute = (unsigned char)SRC_ABS(sw); + s->negate = (unsigned char)SRC_NEG(sw); + if (s->file == TGSI_F_SYSTEM_VALUE) { + if (s->index >= TGSI_PARSE_MAX_IO || + sv_sem[s->index] == 0xff) + return BAD("a system value with no declaration"); + s->index = sv_sem[s->index]; + } + /* The decoded form carries values, not indices, + * so an immediate is resolved here. */ + if (s->file == TGSI_F_IMMEDIATE) { + if (s->index >= nimm) + return TGSI_PARSE_BAD_IMMEDIATE; + memcpy(s->imm, imm[s->index], + sizeof(s->imm)); + } + } + o->nsrc = nsrc; + n++; + i += 1 + nrtok; + continue; + } + + if (type == TOKEN_IMMEDIATE) { + unsigned nrtok = IMM_NRTOK(t); + + if (!nrtok || i + nrtok > ntok) + return TGSI_PARSE_TRUNCATED; + /* one token of header plus four of data */ + if (nrtok != 5) + return TGSI_PARSE_BAD_IMMEDIATE; + if (nimm >= TGSI_PARSE_MAX_IMM) + return TGSI_PARSE_TOO_MANY_IMM; + memcpy(imm[nimm], &tokens[i + 1], 4 * sizeof(float)); + nimm++; + i += nrtok; + continue; + } + + if (type == TOKEN_DECLARATION || type == TOKEN_PROPERTY) { + unsigned nrtok = INS_NRTOK(t); + + /* Declarations carry the binding information a driver + * needs, but the count field is in the same place, so + * they can at least be stepped over correctly. */ + if (!nrtok || i + nrtok > ntok) + return TGSI_PARSE_TRUNCATED; + if (type == TOKEN_DECLARATION && DCL_SEM(t) && + nrtok >= 3) { + unsigned at = i + 2 + DCL_DIM(t) + + DCL_INTERP(t); + + if (at < i + nrtok && DCL_FILE(t) == TGSI_FILE_SV) + sv_note(sv_sem, + DCL_FIRST(tokens[i + 1]), + DCL_LAST(tokens[i + 1]), + SEM_NAME(tokens[at])); + else if (at < i + nrtok && io) + io_note(io, DCL_FILE(t), + DCL_FIRST(tokens[i + 1]), + DCL_LAST(tokens[i + 1]), + SEM_NAME(tokens[at]), + SEM_INDEX(tokens[at])); + } + i += nrtok; + continue; + } + return BAD("a token type this parser does not know"); + } + + *ninsns = n; + return TGSI_PARSE_OK; +} diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/tgsi_parse.h mesa-26.2.2/src/gallium/drivers/sgx/tgsi_parse.h --- mesa-26.2.2.orig/src/gallium/drivers/sgx/tgsi_parse.h 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/tgsi_parse.h 2026-09-08 10:57:36.681924535 +0200 @@ -0,0 +1,72 @@ +/* TGSI token decoding: real Mesa bytecode into the decoded form the translator + * takes. + * + * tgsi_to_uir.c deliberately takes decoded instructions, so the mapping can be + * tested without Mesa. This is the other half: a driver is handed a + * `const struct tgsi_token *`, and something has to turn that into the decoded + * form. Doing it here rather than calling Mesa's parser means the compiler and + * its tests need no Mesa tree, which is the same reason the rest of this + * directory is written the way it is. + * + * The token layout is bit-packed and this decodes it with explicit shifts + * rather than C bit-fields. Bit-field layout is implementation-defined - + * ordering, straddling and signedness all are - and a struct that happens to + * match GCC on x86 is a struct that silently stops matching somewhere else. + * The shifts below are transcribed from Mesa's declarations and each names the + * field it decodes. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: GPL-2.0-only + */ +#ifndef _TGSI_PARSE_H_ +#define _TGSI_PARSE_H_ + +#include + +#include "tgsi_to_uir.h" + +enum tgsi_parse_status { + TGSI_PARSE_OK = 0, + TGSI_PARSE_TRUNCATED, /* a token claims more than is there */ + TGSI_PARSE_BAD_TOKEN, /* a token type with no meaning here */ + TGSI_PARSE_TOO_MANY, /* more instructions than room */ + TGSI_PARSE_TOO_MANY_IMM, + TGSI_PARSE_BAD_IMMEDIATE, /* an immediate that is not 4 floats */ + TGSI_PARSE_BAD_HEADER +}; + +const char *tgsi_parse_status_name(enum tgsi_parse_status st); + +#define TGSI_PARSE_MAX_IMM 64 +#define TGSI_PARSE_MAX_IO 16 + +/* What the declarations say about a shader's inputs and outputs. The decoded + * instruction form carries register indices only, and a driver has to know + * what those indices mean: which output is the position, which varying is a + * texture coordinate. Indexed by register number; 0xff where nothing was + * declared. */ +struct tgsi_io { + uint8_t in_semantic[TGSI_PARSE_MAX_IO]; + uint8_t in_index[TGSI_PARSE_MAX_IO]; + uint8_t out_semantic[TGSI_PARSE_MAX_IO]; + uint8_t out_index[TGSI_PARSE_MAX_IO]; + unsigned nin, nout; +}; + +/* Decode a token stream. Immediates are resolved into the instructions that + * use them, because the decoded form carries values rather than indices. + * *ninsns is set to how many were produced; *stage to 0 for vertex, 1 for + * fragment, matching UIR_STAGE_*. */ +enum tgsi_parse_status tgsi_parse_tokens(const uint32_t *tokens, unsigned ntok, + struct tgsi_insn *out, unsigned cap, + unsigned *ninsns, int *stage); + +/* The same, and the declarations with it. io may be NULL. */ +enum tgsi_parse_status tgsi_parse_tokens_io(const uint32_t *tokens, + unsigned ntok, + struct tgsi_insn *out, unsigned cap, + unsigned *ninsns, int *stage, + struct tgsi_io *io); + +#endif diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/tgsi_to_uir.c mesa-26.2.2/src/gallium/drivers/sgx/tgsi_to_uir.c --- mesa-26.2.2.orig/src/gallium/drivers/sgx/tgsi_to_uir.c 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/tgsi_to_uir.c 2026-09-08 10:57:36.681955596 +0200 @@ -0,0 +1,967 @@ +/* TGSI to UIR - see tgsi_to_uir.h. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: GPL-2.0-only + */ +#include "tgsi_to_uir.h" + +#include + +/* Mesa's numbering, from p_shader_tokens.h. Only the opcodes this maps are + * named here; the test checks each against the generated table so a number + * that moved in Mesa is caught rather than silently mistranslated. */ +#define T_ARL 0 +#define T_MOV 1 +#define T_LIT 2 +#define T_RCP 3 +#define T_RSQ 4 +#define T_MUL 7 +#define T_ADD 8 +#define T_DP3 9 +#define T_DP4 10 +#define T_DP2 71 +#define T_SIN 48 +#define T_COS 36 +#define T_SSG 65 +#define T_DDX 37 +#define T_DDY 38 +#define T_MIN 12 +#define T_MAX 13 +#define T_SLT 14 +#define T_SGE 15 +#define T_MAD 16 +/* 17 is a gap: TGSI_OPCODE_SUB was removed upstream. */ +#define T_END 117 /* aka HALT - the program terminator */ +#define T_LRP 18 +#define T_SQRT 20 +#define T_FRC 24 +#define T_FLR 26 +#define T_EX2 28 +#define T_LG2 29 +#define T_POW 30 +#define T_SEQ 45 +#define T_SGT 47 +#define T_SLE 49 +#define T_SNE 50 +#define T_TEX 52 +#define T_TXP 54 +#define T_DIV 70 +#define T_TRUNC 86 +#define T_CMP 66 +#define T_TXB 68 +#define T_TXL 72 +#define T_KILL_IF 116 +/* control flow */ +#define T_IF 74 +#define T_ELSE 77 +#define T_ENDIF 78 +#define T_BRK 73 +#define T_CONT 96 +#define T_BGNLOOP 99 +#define T_ENDLOOP 101 + +/* The mapping. A TGSI opcode with a direct UIR equivalent is one entry; the + * ones that are not are handled in the switch below. */ +static const struct tgsi_uir_map direct[] = { + { "MOV", T_MOV, UIR_MOV, 1 }, + { "RCP", T_RCP, UIR_RCP, 1 }, + { "RSQ", T_RSQ, UIR_RSQ, 1 }, + { "MUL", T_MUL, UIR_MUL, 2 }, + { "ADD", T_ADD, UIR_ADD, 2 }, + { "DP3", T_DP3, UIR_DP3, 2 }, + { "DP4", T_DP4, UIR_DP4, 2 }, + { "DDX", T_DDX, UIR_DDX, 1 }, + { "DDY", T_DDY, UIR_DDY, 1 }, + { "MIN", T_MIN, UIR_MIN, 2 }, + { "MAX", T_MAX, UIR_MAX, 2 }, + { "SLT", T_SLT, UIR_SLT, 2 }, + { "SGE", T_SGE, UIR_SGE, 2 }, + { "MAD", T_MAD, UIR_MAD, 3 }, + { "FRC", T_FRC, UIR_FRC, 1 }, + { "FLR", T_FLR, UIR_FLOOR, 1 }, + { "EX2", T_EX2, UIR_EXP2, 1 }, + { "LG2", T_LG2, UIR_LOG2, 1 }, + { "SQRT", T_SQRT, UIR_SQRT, 1 }, + { "SEQ", T_SEQ, UIR_SEQ, 2 }, + { "SNE", T_SNE, UIR_SNE, 2 }, + { "CMP", T_CMP, UIR_CMP, 3 }, + /* Sampling. src0 is the coordinate and src1 the sampler, which is the + * order UIR_TEX takes them in; TGSI puts the sampler last for TEX and + * that is the same position with two sources. */ + { "TEX", T_TEX, UIR_TEX, 2 }, + { "TXP", T_TXP, UIR_TEXPROJ, 2 }, + /* TXL and TXB take the level or bias in the coordinate's w and are + * mapped below, where that becomes UIR's third source. */ +}; + +static const unsigned direct_n = sizeof direct / sizeof direct[0]; + +const struct tgsi_uir_map *tgsi_uir_mapping(unsigned *n) +{ + if (n) + *n = direct_n; + return direct; +} + +int tgsi_opcode_supported(unsigned opcode) +{ + unsigned i; + + if (opcode == T_END || opcode == T_TXL || opcode == T_TXB) + return 1; + for (i = 0; i < direct_n; i++) + if (direct[i].tgsi == opcode) + return 1; + return 0; +} + +const char *tgsi_xlat_status_name(enum tgsi_xlat_status st) +{ + switch (st) { + case TGSI_XLAT_OK: return "ok"; + case TGSI_XLAT_UNSUPPORTED_OPCODE: return "unsupported opcode"; + case TGSI_XLAT_UNSUPPORTED_FILE: return "unsupported register file"; + case TGSI_XLAT_BAD_OPERANDS: return "bad operands"; + case TGSI_XLAT_CF_UNBALANCED: return "unbalanced control flow"; + case TGSI_XLAT_CF_NO_LOOP: return "break or continue outside a loop"; + case TGSI_XLAT_CF_NEEDS_CONTEXT: return "control flow needs a context"; + case TGSI_XLAT_UNSUPPORTED_TARGET: return "unsupported texture target"; + } + return "?"; +} + +/* TGSI's register files onto UIR's classes. + * + * TEMPORARY becomes a virtual register because the backend allocates the real + * bank; the fixed classes must not be reallocated, so INPUT, CONSTANT and + * OUTPUT map straight through and keep their index. ADDRESS and SYSTEM_VALUE + * have no place on this hardware and are refused rather than approximated. */ +static int map_file(enum tgsi_file f, enum uir_regclass *cls) +{ + switch (f) { + case TGSI_F_TEMPORARY: *cls = UIR_REG_VIRT; return 0; + case TGSI_F_INPUT: *cls = UIR_REG_IN; return 0; + case TGSI_F_CONSTANT: *cls = UIR_REG_UNIFORM; return 0; + case TGSI_F_OUTPUT: *cls = UIR_REG_OUT; return 0; + case TGSI_F_IMMEDIATE: *cls = UIR_REG_IMM; return 0; + case TGSI_F_SAMPLER: *cls = UIR_REG_SAMPLER; return 0; + default: return -1; + } +} + +/* A TGSI temporary becomes a UIR virtual register with the same index, but the + * shader has to know the register exists: the backend tracks nvirt and refuses + * a virtual it was never told about ("virtual register v0 was never defined"). + * Referencing one without allocating it was the first version of this file. */ +static void ensure_virt(struct uir_shader *s, uint32_t index) +{ + while (s->nvirt <= index) + uir_alloc_virt(s); +} + +static int map_src(struct uir_shader *s, const struct tgsi_src *in, + struct uir_ref *out) +{ + enum uir_regclass cls; + + if (map_file(in->file, &cls)) + return -1; + if (cls == UIR_REG_VIRT) + ensure_virt(s, in->index); + memset(out, 0, sizeof(*out)); + out->cls = cls; + out->index = in->index; + /* No sentinel here. TGSI always carries an explicit swizzle, and 0x00 + * is .xxxx - a legitimate one. Treating zero as "unset" and + * substituting identity silently turned .xxxx into .xyzw, which + * compiles and renders and is wrong. */ + out->swizzle = in->swizzle; + out->negate = in->negate; + out->absolute = in->absolute; + if (cls == UIR_REG_IMM) + memcpy(out->imm, in->imm, sizeof(out->imm)); + return 0; +} + +static int map_dst(struct uir_shader *s, const struct tgsi_dst *in, + struct uir_ref *out) +{ + enum uir_regclass cls; + + if (map_file(in->file, &cls)) + return -1; + if (cls == UIR_REG_VIRT) + ensure_virt(s, in->index); + /* Only these three can be written. A shader writing a constant or a + * sampler is malformed, and letting it through would produce code that + * writes a bank the PDS is loading. */ + if (cls != UIR_REG_VIRT && cls != UIR_REG_OUT && cls != UIR_REG_IN) + return -1; + memset(out, 0, sizeof(*out)); + out->cls = cls; + out->index = in->index; + out->swizzle = 0xe4; + return 0; +} + +/* IF branches *past* its body when the condition is false, while UIR_BRC + * branches when src0 is non-zero. Negating the source does not bridge that: + * the negative of a non-zero value is still non-zero, so a negated BRC still + * branches into the body. The inversion is done with the control flow itself + * instead - branch into the body when the condition holds, and fall through + * to an unconditional jump over it when it does not - which needs neither a + * comparison nor a temporary to hold one. */ +static enum tgsi_xlat_status xlat_cf(struct uir_shader *s, + struct tgsi_xlat_ctx *ctx, + const struct tgsi_insn *in) +{ + struct uir_insn *b; + + if (!ctx) + return TGSI_XLAT_CF_NEEDS_CONTEXT; + + switch (in->opcode) { + case T_IF: { + uint32_t body; + + if (in->nsrc != 1) + return TGSI_XLAT_BAD_OPERANDS; + if (ctx->depth >= TGSI_MAX_CF_DEPTH) + return TGSI_XLAT_CF_UNBALANCED; + ctx->cf[ctx->depth].kind = TGSI_CF_IF; + ctx->cf[ctx->depth].a = uir_alloc_label(s); + ctx->cf[ctx->depth].b = uir_alloc_label(s); + ctx->cf[ctx->depth].has_else = 0; + body = uir_alloc_label(s); + + b = uir_emit(s, UIR_BRC); + if (!b) + return TGSI_XLAT_BAD_OPERANDS; + b->type = UIR_F32; + b->writemask = 0xf; + if (map_src(s, &in->src[0], &b->src[0])) + return TGSI_XLAT_UNSUPPORTED_FILE; + b->nsrc = 1; + b->branch_target = body; + + b = uir_emit(s, UIR_BR); + if (!b) + return TGSI_XLAT_BAD_OPERANDS; + b->branch_target = ctx->cf[ctx->depth].a; + + b = uir_emit(s, UIR_LABEL); + if (!b) + return TGSI_XLAT_BAD_OPERANDS; + b->label_id = body; + ctx->depth++; + return TGSI_XLAT_OK; + } + + case T_ELSE: { + struct uir_insn *lbl; + + if (!ctx->depth || + ctx->cf[ctx->depth - 1].kind != TGSI_CF_IF) + return TGSI_XLAT_CF_UNBALANCED; + if (ctx->cf[ctx->depth - 1].has_else) + return TGSI_XLAT_CF_UNBALANCED; + /* the true body falls out over the else body */ + b = uir_emit(s, UIR_BR); + if (!b) + return TGSI_XLAT_BAD_OPERANDS; + b->branch_target = ctx->cf[ctx->depth - 1].b; + lbl = uir_emit(s, UIR_LABEL); + if (!lbl) + return TGSI_XLAT_BAD_OPERANDS; + lbl->label_id = ctx->cf[ctx->depth - 1].a; + ctx->cf[ctx->depth - 1].has_else = 1; + return TGSI_XLAT_OK; + } + + case T_ENDIF: { + struct uir_insn *lbl; + + if (!ctx->depth || + ctx->cf[ctx->depth - 1].kind != TGSI_CF_IF) + return TGSI_XLAT_CF_UNBALANCED; + ctx->depth--; + /* Without an ELSE the false path jumps straight here, so the + * else label has to land somewhere: both labels go at the end. + */ + if (!ctx->cf[ctx->depth].has_else) { + lbl = uir_emit(s, UIR_LABEL); + if (!lbl) + return TGSI_XLAT_BAD_OPERANDS; + lbl->label_id = ctx->cf[ctx->depth].a; + } + lbl = uir_emit(s, UIR_LABEL); + if (!lbl) + return TGSI_XLAT_BAD_OPERANDS; + lbl->label_id = ctx->cf[ctx->depth].b; + return TGSI_XLAT_OK; + } + + case T_BGNLOOP: { + struct uir_insn *lbl; + + if (ctx->depth >= TGSI_MAX_CF_DEPTH) + return TGSI_XLAT_CF_UNBALANCED; + ctx->cf[ctx->depth].kind = TGSI_CF_LOOP; + ctx->cf[ctx->depth].a = uir_alloc_label(s); /* top */ + ctx->cf[ctx->depth].b = uir_alloc_label(s); /* break */ + ctx->cf[ctx->depth].has_else = 0; + lbl = uir_emit(s, UIR_LABEL); + if (!lbl) + return TGSI_XLAT_BAD_OPERANDS; + lbl->label_id = ctx->cf[ctx->depth].a; + ctx->depth++; + return TGSI_XLAT_OK; + } + + case T_ENDLOOP: { + struct uir_insn *lbl; + + if (!ctx->depth || + ctx->cf[ctx->depth - 1].kind != TGSI_CF_LOOP) + return TGSI_XLAT_CF_UNBALANCED; + ctx->depth--; + b = uir_emit(s, UIR_BR); /* back to the top */ + if (!b) + return TGSI_XLAT_BAD_OPERANDS; + b->branch_target = ctx->cf[ctx->depth].a; + lbl = uir_emit(s, UIR_LABEL); /* where BRK lands */ + if (!lbl) + return TGSI_XLAT_BAD_OPERANDS; + lbl->label_id = ctx->cf[ctx->depth].b; + return TGSI_XLAT_OK; + } + + case T_BRK: + case T_CONT: { + unsigned d = ctx->depth; + + /* Walk out to the innermost enclosing loop, past any IFs. + * Targeting ctx->cf[depth - 1] blindly breaks the IF instead, + * which is a shader that loops the wrong number of times. */ + while (d && ctx->cf[d - 1].kind != TGSI_CF_LOOP) + d--; + if (!d) + return TGSI_XLAT_CF_NO_LOOP; + b = uir_emit(s, UIR_BR); + if (!b) + return TGSI_XLAT_BAD_OPERANDS; + b->branch_target = (in->opcode == T_BRK) + ? ctx->cf[d - 1].b /* out of the loop */ + : ctx->cf[d - 1].a; /* back to the top */ + return TGSI_XLAT_OK; + } + + default: + return TGSI_XLAT_UNSUPPORTED_OPCODE; + } +} + +enum tgsi_xlat_status tgsi_insn_to_uir(struct uir_shader *s, + const struct tgsi_insn *in) +{ + return tgsi_insn_to_uir_ctx(s, NULL, in); +} + +/* A system value has no register the instruction could read, so it is + * computed into a virtual register just ahead of its user and the instruction + * reads that instead. Returns 1 with *copy rewritten, 0 when there was nothing + * to do, -1 for a value this hardware cannot produce. */ +static int lower_sysvals(struct uir_shader *s, const struct tgsi_insn *in, + struct tgsi_insn *copy) +{ + unsigned j; + int done = 0; + + for (j = 0; j < in->nsrc; j++) { + struct uir_insn *f; + uint32_t v; + + if (in->src[j].file != TGSI_F_SYSTEM_VALUE) + continue; + if (in->src[j].index != TGSI_SEM_FACE) + return -1; + if (!done) + *copy = *in; + v = uir_alloc_virt(s); + f = uir_emit(s, UIR_FACE); + if (!f) + return -1; + f->type = UIR_F32; + f->writemask = 0xf; + f->dst = uir_reg(UIR_REG_VIRT, v); + f->nsrc = 0; + copy->src[j].file = TGSI_F_TEMPORARY; + copy->src[j].index = v; + done = 1; + } + return done; +} + +enum tgsi_xlat_status tgsi_insn_to_uir_ctx(struct uir_shader *s, + struct tgsi_xlat_ctx *ctx, + const struct tgsi_insn *in) +{ + struct uir_insn *out; + struct tgsi_insn sv; + unsigned i, j; + int lowered; + + if (!s || !in) + return TGSI_XLAT_BAD_OPERANDS; + + lowered = lower_sysvals(s, in, &sv); + if (lowered < 0) + return TGSI_XLAT_UNSUPPORTED_FILE; + if (lowered) + in = &sv; + + switch (in->opcode) { + case T_IF: case T_ELSE: case T_ENDIF: + case T_BGNLOOP: case T_ENDLOOP: case T_BRK: case T_CONT: + return xlat_cf(s, ctx, in); + /* The part samples 2D surfaces, volumes and cubes: a 1D one is a + * row of a 2D texture (lowered by the driver), a rectangle a 2D one + * with unnormalised coordinates the driver scales, and a volume or + * cube a three-coordinate sample. A shadow target's comparison is not in + * the unit - the DDK's CMPMODE is a fourth descriptor word this + * core does not have - so the driver lowers it into the program + * before the tokens get here (nir_lower_tex_shadow), and one that + * still arrives would sample the depth as a colour. Volumes, + * arrays and multisampled targets have no descriptor at all. */ + case T_TEX: case T_TXP: case T_TXL: case T_TXB: + switch (in->tex_target) { + /* BUFFER is Mesa's zero, which is also what a decoded + * instruction built without a texture token carries; no + * buffer texture can reach here, since the screen does not + * offer them, so it reads as "unspecified". A volume and a + * cube read a third coordinate (tex_dim below). */ + case TGSI_TEX_BUFFER: case TGSI_TEX_1D: case TGSI_TEX_2D: + case TGSI_TEX_3D: case TGSI_TEX_RECT: case TGSI_TEX_CUBE: + break; + default: + return TGSI_XLAT_UNSUPPORTED_TARGET; + } + break; + default: + break; + } + + for (i = 0; i < direct_n; i++) { + if (direct[i].tgsi != in->opcode) + continue; + if (in->nsrc != direct[i].nsrc) + return TGSI_XLAT_BAD_OPERANDS; + + out = uir_emit(s, (enum uir_op)direct[i].uir); + if (!out) + return TGSI_XLAT_BAD_OPERANDS; + out->type = UIR_F32; + out->saturate = in->saturate; + out->writemask = in->dst.writemask ? in->dst.writemask : 0xf; + if (map_dst(s, &in->dst, &out->dst)) + return TGSI_XLAT_UNSUPPORTED_FILE; + for (j = 0; j < in->nsrc; j++) + if (map_src(s, &in->src[j], &out->src[j])) + return TGSI_XLAT_UNSUPPORTED_FILE; + out->nsrc = in->nsrc; + if (out->op == UIR_TEX || out->op == UIR_TEXPROJ || + out->op == UIR_TEXLOD) + out->tex_dim = (uint8_t)tgsi_tex_coord_dim(in->tex_target); + return TGSI_XLAT_OK; + } + + /* TXL samples at the level in the coordinate's w, TXB at the level + * the unit computes plus that w. UIR carries the value as a third + * source of its own, so it is the coordinate again, read through w. */ + if (in->opcode == T_TXL || in->opcode == T_TXB) { + unsigned w; + + /* Two sources as Mesa writes it; three when the driver has + * folded the coordinate onto its varying and moved the value + * out of the temporary's w into a source of its own. */ + if (in->nsrc != 2 && in->nsrc != 3) + return TGSI_XLAT_BAD_OPERANDS; + out = uir_emit(s, in->opcode == T_TXL ? UIR_TEXLOD : UIR_TEXBIAS); + if (!out) + return TGSI_XLAT_BAD_OPERANDS; + out->type = UIR_F32; + out->saturate = in->saturate; + out->writemask = in->dst.writemask ? in->dst.writemask : 0xf; + if (map_dst(s, &in->dst, &out->dst) || + map_src(s, &in->src[0], &out->src[0]) || + map_src(s, &in->src[1], &out->src[1]) || + map_src(s, &in->src[in->nsrc == 3 ? 2 : 0], &out->src[2])) + return TGSI_XLAT_UNSUPPORTED_FILE; + if (in->nsrc == 2) { + w = (in->src[0].swizzle >> 6) & 3u; + out->src[2].swizzle = (unsigned char)(w | (w << 2) | + (w << 4) | (w << 6)); + } + out->nsrc = 3; + out->tex_dim = (uint8_t)tgsi_tex_coord_dim(in->tex_target); + return TGSI_XLAT_OK; + } + + /* TGSI has both directions of every comparison; UIR has SLT and SGE. + * SGT(a,b) is SLT(b,a) and SLE(a,b) is SGE(b,a), so they are the same + * instruction with the sources exchanged rather than two more opcodes + * for the backend to lower. */ + if (in->opcode == T_SGT || in->opcode == T_SLE) { + struct tgsi_insn t = *in; + + if (in->nsrc != 2) + return TGSI_XLAT_BAD_OPERANDS; + t.opcode = (in->opcode == T_SGT) ? T_SLT : T_SGE; + t.src[0] = in->src[1]; + t.src[1] = in->src[0]; + return tgsi_insn_to_uir_ctx(s, ctx, &t); + } + + /* SSG is sign(x): -1, 0 or +1. Two comparisons and a subtract, which + * gets the zero case right where a single SGE against zero does not - + * SGE(0, 0) is one, and sign(0) is zero. */ + if (in->opcode == T_SSG) { + struct uir_insn *pos, *neg, *sub; + uint32_t p1 = uir_alloc_virt(s); + uint32_t n1 = uir_alloc_virt(s); + struct uir_ref zero, pr, nr; + + if (in->nsrc != 1) + return TGSI_XLAT_BAD_OPERANDS; + zero = uir_imm1(0.0f); + memset(&pr, 0, sizeof pr); + pr.cls = UIR_REG_VIRT; + pr.index = p1; + pr.swizzle = 0xe4; + nr = pr; + nr.index = n1; + + /* 1 where 0 < x */ + pos = uir_emit(s, UIR_SLT); + if (!pos) + return TGSI_XLAT_BAD_OPERANDS; + pos->type = UIR_F32; + pos->writemask = in->dst.writemask; + pos->dst.cls = UIR_REG_VIRT; + pos->dst.index = p1; + pos->dst.swizzle = 0xe4; + pos->src[0] = zero; + if (map_src(s, &in->src[0], &pos->src[1])) + return TGSI_XLAT_UNSUPPORTED_FILE; + pos->nsrc = 2; + + /* 1 where x < 0 */ + neg = uir_emit(s, UIR_SLT); + if (!neg) + return TGSI_XLAT_BAD_OPERANDS; + neg->type = UIR_F32; + neg->writemask = in->dst.writemask; + neg->dst.cls = UIR_REG_VIRT; + neg->dst.index = n1; + neg->dst.swizzle = 0xe4; + if (map_src(s, &in->src[0], &neg->src[0])) + return TGSI_XLAT_UNSUPPORTED_FILE; + neg->src[1] = zero; + neg->nsrc = 2; + + sub = uir_emit(s, UIR_ADD); + if (!sub) + return TGSI_XLAT_BAD_OPERANDS; + sub->type = UIR_F32; + if (map_dst(s, &in->dst, &sub->dst)) + return TGSI_XLAT_UNSUPPORTED_FILE; + sub->writemask = in->dst.writemask; + sub->src[0] = pr; + sub->src[1] = nr; + sub->src[1].negate = 1; + sub->nsrc = 2; + return TGSI_XLAT_OK; + } + + /* SIN and COS have no instruction at all: the transcendental unit is + * RCP, RSQ, LOG, EXP and FRC, and the seventy-five USSE opcodes have + * no trig. So they are lowered the way every driver on hardware + * without trig lowers them - range reduction and a polynomial. + * + * Reduce with FRC: x/2pi + bias, take the fraction, scale back. The + * bias is 0.5 for sine and 0.75 for cosine, because cos(x) is + * sin(x + pi/2) and pi/2 is a quarter turn - so the two differ by one + * constant and share everything else. + * + * Then the usual degree-two-plus-correction fit, which is accurate to + * about a thousandth over the whole circle: + * y = x * (B + C*|x|), B = 4/pi, C = -4/pi^2 + * r = P * (y*|y| - y) + y, P = 0.225 + * + * TGSI's SIN and COS are scalar: the source's x is used and the result + * replicated across the destination mask, which is what the swizzles + * below do. */ + if (in->opcode == T_SIN || in->opcode == T_COS) { + static const float k_inv2pi = 0.15915494309189535f; + static const float k_2pi = 6.28318530717958648f; + static const float k_pi = 3.14159265358979324f; + static const float k_b = 1.27323954473516269f; + static const float k_c = -0.40528473456935109f; + static const float k_p = 0.225f; + struct uir_insn *i0, *i1, *i2, *i3, *i4, *i5, *i6, *i7; + uint32_t t = uir_alloc_virt(s); + uint32_t y = uir_alloc_virt(s); + struct uir_ref tx, yx; + + if (in->nsrc != 1) + return TGSI_XLAT_BAD_OPERANDS; + memset(&tx, 0, sizeof tx); + tx.cls = UIR_REG_VIRT; + tx.index = t; + tx.swizzle = UIR_SWZ_XXXX; + yx = tx; + yx.index = y; + +#define SIN_SCALAR(ins, op) \ + do { \ + (ins) = uir_emit(s, (op)); \ + if (!(ins)) \ + return TGSI_XLAT_BAD_OPERANDS; \ + (ins)->type = UIR_F32; \ + (ins)->writemask = 0x1; \ + (ins)->dst.cls = UIR_REG_VIRT; \ + (ins)->dst.swizzle = 0xe4; \ + } while (0) + + /* t = frac(x * 1/2pi + bias) */ + SIN_SCALAR(i0, UIR_MAD); + i0->dst.index = t; + if (map_src(s, &in->src[0], &i0->src[0])) + return TGSI_XLAT_UNSUPPORTED_FILE; + i0->src[1] = uir_imm1(k_inv2pi); + i0->src[2] = uir_imm1(in->opcode == T_COS ? 0.75f : 0.5f); + i0->nsrc = 3; + + SIN_SCALAR(i1, UIR_FRC); + i1->dst.index = t; + i1->src[0] = tx; + i1->nsrc = 1; + + /* x' = t * 2pi - pi, in [-pi, pi) */ + SIN_SCALAR(i2, UIR_MAD); + i2->dst.index = t; + i2->src[0] = tx; + i2->src[1] = uir_imm1(k_2pi); + i2->src[2] = uir_imm1(-k_pi); + i2->nsrc = 3; + + /* y = x' * (C*|x'| + B) */ + SIN_SCALAR(i3, UIR_MAD); + i3->dst.index = y; + i3->src[0] = tx; + i3->src[0].absolute = 1; + i3->src[1] = uir_imm1(k_c); + i3->src[2] = uir_imm1(k_b); + i3->nsrc = 3; + + SIN_SCALAR(i4, UIR_MUL); + i4->dst.index = y; + i4->src[0] = tx; + i4->src[1] = yx; + i4->nsrc = 2; + + /* r = P * (y*|y| - y) + y */ + SIN_SCALAR(i5, UIR_MUL); + i5->dst.index = t; + i5->src[0] = yx; + i5->src[1] = yx; + i5->src[1].absolute = 1; + i5->nsrc = 2; + + SIN_SCALAR(i6, UIR_ADD); + i6->dst.index = t; + i6->src[0] = tx; + i6->src[1] = yx; + i6->src[1].negate = 1; + i6->nsrc = 2; +#undef SIN_SCALAR + + i7 = uir_emit(s, UIR_MAD); + if (!i7) + return TGSI_XLAT_BAD_OPERANDS; + i7->type = UIR_F32; + if (map_dst(s, &in->dst, &i7->dst)) + return TGSI_XLAT_UNSUPPORTED_FILE; + i7->writemask = in->dst.writemask; + i7->src[0] = tx; + i7->src[1] = uir_imm1(k_p); + i7->src[2] = yx; + i7->nsrc = 3; + return TGSI_XLAT_OK; + } + + /* DP2 has no instruction of its own either, and needs no transcendental + * to stand in: a.x*b.x + a.y*b.y is a two-channel multiply and one add + * of the two channels against each other, replicated the way the three + * and four-component dots already are. ioquake3's opengl2 renderer + * refuses to link without it. */ + if (in->opcode == T_DP2) { + struct uir_insn *mul, *add; + uint32_t t = uir_alloc_virt(s); + + if (in->nsrc != 2) + return TGSI_XLAT_BAD_OPERANDS; + mul = uir_emit(s, UIR_MUL); + if (!mul) + return TGSI_XLAT_BAD_OPERANDS; + mul->type = UIR_F32; + mul->writemask = 0x3; /* x and y */ + mul->dst.cls = UIR_REG_VIRT; + mul->dst.index = t; + mul->dst.swizzle = 0xe4; + if (map_src(s, &in->src[0], &mul->src[0]) || + map_src(s, &in->src[1], &mul->src[1])) + return TGSI_XLAT_UNSUPPORTED_FILE; + mul->nsrc = 2; + + add = uir_emit(s, UIR_ADD); + if (!add) + return TGSI_XLAT_BAD_OPERANDS; + add->type = UIR_F32; + if (map_dst(s, &in->dst, &add->dst)) + return TGSI_XLAT_UNSUPPORTED_FILE; + add->writemask = in->dst.writemask; + memset(&add->src[0], 0, sizeof(add->src[0])); + add->src[0].cls = UIR_REG_VIRT; + add->src[0].index = t; + add->src[0].swizzle = 0x00; /* .xxxx */ + memset(&add->src[1], 0, sizeof(add->src[1])); + add->src[1].cls = UIR_REG_VIRT; + add->src[1].index = t; + add->src[1].swizzle = 0x55; /* .yyyy */ + add->nsrc = 2; + return TGSI_XLAT_OK; + } + + /* POW has no instruction of its own: a^b is exp2(b * log2(a)), which + * the part already has both halves of. Both are scalar, so the + * intermediate is one channel of a fresh virtual register broadcast + * back out. glmark2's phong and blinn-phong shading both need it, and + * without it the whole fragment program was refused. */ + if (in->opcode == T_POW) { + struct uir_insn *lg, *mul, *ex; + uint32_t t = uir_alloc_virt(s); + + if (in->nsrc != 2) + return TGSI_XLAT_BAD_OPERANDS; + lg = uir_emit(s, UIR_LOG2); + if (!lg) + return TGSI_XLAT_BAD_OPERANDS; + lg->type = UIR_F32; + lg->writemask = 0x1; + lg->dst.cls = UIR_REG_VIRT; + lg->dst.index = t; + lg->dst.swizzle = 0xe4; + if (map_src(s, &in->src[0], &lg->src[0])) + return TGSI_XLAT_UNSUPPORTED_FILE; + lg->nsrc = 1; + + mul = uir_emit(s, UIR_MUL); + if (!mul) + return TGSI_XLAT_BAD_OPERANDS; + mul->type = UIR_F32; + mul->writemask = 0x1; + mul->dst = lg->dst; + if (map_src(s, &in->src[1], &mul->src[0])) + return TGSI_XLAT_UNSUPPORTED_FILE; + memset(&mul->src[1], 0, sizeof(mul->src[1])); + mul->src[1].cls = UIR_REG_VIRT; + mul->src[1].index = t; + mul->src[1].swizzle = 0x00; /* .xxxx */ + mul->nsrc = 2; + + ex = uir_emit(s, UIR_EXP2); + if (!ex) + return TGSI_XLAT_BAD_OPERANDS; + ex->type = UIR_F32; + if (map_dst(s, &in->dst, &ex->dst)) + return TGSI_XLAT_UNSUPPORTED_FILE; + ex->writemask = in->dst.writemask; + memset(&ex->src[0], 0, sizeof(ex->src[0])); + ex->src[0].cls = UIR_REG_VIRT; + ex->src[0].index = t; + ex->src[0].swizzle = 0x00; + ex->nsrc = 1; + return TGSI_XLAT_OK; + } + + /* DIV has no instruction of its own: a/b is a * (1/b), and the + * reciprocal is per component here, so the whole thing is two + * instructions under the destination's own write mask. sway's + * clients want it - alacritty's fragment program is refused without + * it and the terminal never comes up. */ + if (in->opcode == T_DIV) { + struct uir_insn *rc, *mul; + uint32_t t = uir_alloc_virt(s); + + if (in->nsrc != 2) + return TGSI_XLAT_BAD_OPERANDS; + rc = uir_emit(s, UIR_RCP); + if (!rc) + return TGSI_XLAT_BAD_OPERANDS; + rc->type = UIR_F32; + rc->writemask = in->dst.writemask; + rc->dst.cls = UIR_REG_VIRT; + rc->dst.index = t; + rc->dst.swizzle = 0xe4; + if (map_src(s, &in->src[1], &rc->src[0])) + return TGSI_XLAT_UNSUPPORTED_FILE; + rc->nsrc = 1; + + mul = uir_emit(s, UIR_MUL); + if (!mul) + return TGSI_XLAT_BAD_OPERANDS; + mul->type = UIR_F32; + if (map_dst(s, &in->dst, &mul->dst)) + return TGSI_XLAT_UNSUPPORTED_FILE; + mul->writemask = in->dst.writemask; + if (map_src(s, &in->src[0], &mul->src[0])) + return TGSI_XLAT_UNSUPPORTED_FILE; + memset(&mul->src[1], 0, sizeof(mul->src[1])); + mul->src[1].cls = UIR_REG_VIRT; + mul->src[1].index = t; + mul->src[1].swizzle = 0xe4; + mul->nsrc = 2; + return TGSI_XLAT_OK; + } + + /* TRUNC rounds towards zero, which floor does not: floor(-1.5) is -2 + * where trunc(-1.5) is -1. So take the floor of the magnitude and put + * the sign back, which is what the select is for. */ + if (in->opcode == T_TRUNC) { + struct uir_insn *fl, *cm; + uint32_t t = uir_alloc_virt(s); + + if (in->nsrc != 1) + return TGSI_XLAT_BAD_OPERANDS; + fl = uir_emit(s, UIR_FLOOR); + if (!fl) + return TGSI_XLAT_BAD_OPERANDS; + fl->type = UIR_F32; + fl->writemask = in->dst.writemask; + fl->dst.cls = UIR_REG_VIRT; + fl->dst.index = t; + fl->dst.swizzle = 0xe4; + if (map_src(s, &in->src[0], &fl->src[0])) + return TGSI_XLAT_UNSUPPORTED_FILE; + fl->src[0].absolute = 1; + fl->src[0].negate = 0; + fl->nsrc = 1; + + /* src0 < 0 ? -floor(|src0|) : floor(|src0|) */ + cm = uir_emit(s, UIR_CMP); + if (!cm) + return TGSI_XLAT_BAD_OPERANDS; + cm->type = UIR_F32; + if (map_dst(s, &in->dst, &cm->dst)) + return TGSI_XLAT_UNSUPPORTED_FILE; + cm->writemask = in->dst.writemask; + if (map_src(s, &in->src[0], &cm->src[0])) + return TGSI_XLAT_UNSUPPORTED_FILE; + memset(&cm->src[1], 0, sizeof(cm->src[1])); + cm->src[1].cls = UIR_REG_VIRT; + cm->src[1].index = t; + cm->src[1].swizzle = 0xe4; + cm->src[1].negate = 1; + memset(&cm->src[2], 0, sizeof(cm->src[2])); + cm->src[2].cls = UIR_REG_VIRT; + cm->src[2].index = t; + cm->src[2].swizzle = 0xe4; + cm->nsrc = 3; + return TGSI_XLAT_OK; + } + + /* KILL_IF discards where src0 < 0, which is UIR_KILL exactly. The + * shader has to say so: the backend keeps the depth write until after + * the kill, and cannot know to do that from the instruction alone. */ + if (in->opcode == T_KILL_IF) { + struct uir_insn *k; + + if (in->nsrc != 1) + return TGSI_XLAT_BAD_OPERANDS; + k = uir_emit(s, UIR_KILL); + if (!k) + return TGSI_XLAT_BAD_OPERANDS; + k->type = UIR_F32; + k->writemask = 0xf; + if (map_src(s, &in->src[0], &k->src[0])) + return TGSI_XLAT_UNSUPPORTED_FILE; + k->nsrc = 1; + s->uses_kill = 1; + return TGSI_XLAT_OK; + } + + /* END terminates the program. It becomes UIR_EMIT, which is what the + * backend turns into the output write and the .end flag on the last + * instruction - and a USSE program whose last instruction lacks .end + * runs on into whatever follows it in memory. */ + if (in->opcode == T_END) { + struct uir_insn *out2 = uir_emit(s, UIR_EMIT); + + if (!out2) + return TGSI_XLAT_BAD_OPERANDS; + out2->type = UIR_F32; + out2->writemask = 0xf; + return TGSI_XLAT_OK; + } + + /* There is deliberately no SUB case. TGSI_OPCODE_SUB was removed + * upstream - 17 is a gap in Mesa 26.2.1 - and Mesa lowers subtraction + * to ADD with a negated source before a driver ever sees it. A special + * case here would be dead code keyed to an opcode that no longer + * exists, which is what the first version of this file was. */ + return TGSI_XLAT_UNSUPPORTED_OPCODE; +} + +enum tgsi_xlat_status tgsi_to_uir(struct uir_shader *s, + const struct tgsi_insn *insns, unsigned n, + unsigned *at) +{ + struct tgsi_xlat_ctx ctx; + unsigned i; + + memset(&ctx, 0, sizeof(ctx)); + /* Claim every temporary the shader names before translating, so a + * scratch register taken for a lowered opcode sits above them all. + * TGSI temporaries map onto virtual registers by index, so one taken + * from the middle is the same register as a TEMP the shader has not + * mentioned yet. */ + for (i = 0; i < n; i++) { + unsigned j; + + if (insns[i].dst.file == TGSI_F_TEMPORARY) + ensure_virt(s, insns[i].dst.index); + for (j = 0; j < insns[i].nsrc; j++) + if (insns[i].src[j].file == TGSI_F_TEMPORARY) + ensure_virt(s, insns[i].src[j].index); + } + for (i = 0; i < n; i++) { + enum tgsi_xlat_status st = + tgsi_insn_to_uir_ctx(s, &ctx, &insns[i]); + + if (st != TGSI_XLAT_OK) { + if (at) + *at = i; + return st; + } + } + /* An IF left open would produce a branch to a label that is never + * emitted, which the backend resolves to whatever follows. */ + if (ctx.depth) { + if (at) + *at = n; + return TGSI_XLAT_CF_UNBALANCED; + } + return TGSI_XLAT_OK; +} diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/tgsi_to_uir.h mesa-26.2.2/src/gallium/drivers/sgx/tgsi_to_uir.h --- mesa-26.2.2.orig/src/gallium/drivers/sgx/tgsi_to_uir.h 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/tgsi_to_uir.h 2026-09-08 10:57:36.681975118 +0200 @@ -0,0 +1,173 @@ +/* TGSI to UIR - the front end a Gallium driver needs. + * + * usse-cc compiles GLSL, which is not what a Gallium driver is handed. Gallium + * hands a driver TGSI (or NIR, which Mesa can lower to TGSI), and the backend + * that turns UIR into SGX535 machine code already exists and already runs on + * silicon. So the missing link is exactly this: TGSI in, UIR out. + * + * The input is decoded TGSI rather than raw tokens. Token parsing is Mesa's + * job and its result is a struct much like this one; keeping them separate + * means this file is about the *mapping* - which is the part that needs the + * hardware knowledge - and is testable without Mesa in the loop. + * + * Opcode numbers come from Mesa's own p_shader_tokens.h, extracted by + * extract-opcodes.sh rather than transcribed. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: GPL-2.0-only + */ +#ifndef _TGSI_TO_UIR_H_ +#define _TGSI_TO_UIR_H_ + +#include "usse_ir.h" + +/* TGSI_FILE_*, in Mesa's numbering. */ +enum tgsi_file { + TGSI_F_NULL = 0, + TGSI_F_CONSTANT, + TGSI_F_INPUT, + TGSI_F_OUTPUT, + TGSI_F_TEMPORARY, + TGSI_F_SAMPLER, + TGSI_F_ADDRESS, + TGSI_F_IMMEDIATE, + TGSI_F_SYSTEM_VALUE +}; + +/* A SYSTEM_VALUE operand is named by its TGSI semantic, not by its + * declaration index: tgsi_parse.c rewrites the index from the declaration, and + * a driver that synthesises one writes the semantic directly. FACE is the only + * one this hardware has a source for - the ISP's backface bit - and it comes + * out as +1.0 for a front-facing fragment and -1.0 otherwise, which is the + * float form Mesa's FACE input takes. The number is Mesa's TGSI_SEMANTIC_FACE. */ +#define TGSI_SEM_FACE 7 + +struct tgsi_src { + enum tgsi_file file; + unsigned index; + unsigned char swizzle; /* 2 bits per channel, 0xe4 identity */ + unsigned char negate, absolute; + float imm[4]; /* when file is TGSI_F_IMMEDIATE */ +}; + +struct tgsi_dst { + enum tgsi_file file; + unsigned index; + unsigned char writemask; /* bit0 = x .. bit3 = w */ +}; + +/* TGSI_TEXTURE_*, in Mesa's numbering (p_shader_tokens.h). */ +enum tgsi_tex_target { + TGSI_TEX_BUFFER = 0, + TGSI_TEX_1D, + TGSI_TEX_2D, + TGSI_TEX_3D, + TGSI_TEX_CUBE, + TGSI_TEX_RECT, + TGSI_TEX_SHADOW1D, + TGSI_TEX_SHADOW2D, + TGSI_TEX_SHADOWRECT, + TGSI_TEX_1D_ARRAY, + TGSI_TEX_2D_ARRAY, + TGSI_TEX_SHADOW1D_ARRAY, + TGSI_TEX_SHADOW2D_ARRAY, + TGSI_TEX_SHADOWCUBE, + TGSI_TEX_2D_MSAA, + TGSI_TEX_2D_ARRAY_MSAA, + TGSI_TEX_CUBE_ARRAY, + TGSI_TEX_SHADOWCUBE_ARRAY +}; + +struct tgsi_insn { + unsigned opcode; /* TGSI_OPCODE_*, Mesa's numbering */ + unsigned char saturate; + /* What a sampling instruction samples, enum tgsi_tex_target, from + * the texture token of a TEX/TXP/TXL/TXB; zero (BUFFER) on anything + * else. A volume or cube sample reads three coordinate components; + * the translator refuses what the part has no descriptor for and + * what the driver has to lower first - a shadow target's comparison. */ + unsigned char tex_target; + struct tgsi_dst dst; + struct tgsi_src src[4]; + unsigned nsrc; +}; + +/* Coordinate components a sample of the target reads: three for a volume + * or a cube, two otherwise. */ +static inline unsigned tgsi_tex_coord_dim(unsigned char target) +{ + return target == TGSI_TEX_3D || target == TGSI_TEX_CUBE || + target == TGSI_TEX_SHADOWCUBE || target == TGSI_TEX_CUBE_ARRAY + ? 3u : 2u; +} + +/* Why a TGSI instruction could not be translated. */ +enum tgsi_xlat_status { + TGSI_XLAT_OK = 0, + TGSI_XLAT_UNSUPPORTED_OPCODE, + TGSI_XLAT_UNSUPPORTED_FILE, + TGSI_XLAT_BAD_OPERANDS, + TGSI_XLAT_CF_UNBALANCED, /* ELSE or ENDIF with no IF, or too deep */ + TGSI_XLAT_CF_NO_LOOP, /* BRK or CONT outside any loop */ + TGSI_XLAT_CF_NEEDS_CONTEXT, /* control flow without a ctx */ + TGSI_XLAT_UNSUPPORTED_TARGET /* a texture target not sampled here */ +}; + +/* Control flow is not per-instruction: IF needs to know where its ELSE and + * ENDIF will land, so the labels outlive the instruction that opened them. + * This is that state. Depth is bounded because a shader that nests deeper than + * this has other problems, and running off the end is reported rather than + * scribbling. */ +#define TGSI_MAX_CF_DEPTH 16 + +/* One stack for both constructs, with a kind, because BRK inside an IF inside + * a loop has to leave the *loop*. A separate if-stack and loop-stack gets that + * right only by accident, and a BRK that breaks the wrong block is a shader + * that runs the wrong number of times rather than one that fails to build. */ +enum tgsi_cf_kind { TGSI_CF_IF, TGSI_CF_LOOP }; + +struct tgsi_xlat_ctx { + struct { + enum tgsi_cf_kind kind; + uint32_t a, b; /* IF: else, endif. LOOP: top, break */ + int has_else; + } cf[TGSI_MAX_CF_DEPTH]; + unsigned depth; +}; + +/* Translate one instruction, appending to the shader. Some TGSI opcodes expand + * to more than one UIR instruction, so this may emit several. ctx may be NULL, + * in which case control-flow opcodes are refused rather than mistranslated - + * they cannot be handled without it. */ +enum tgsi_xlat_status tgsi_insn_to_uir_ctx(struct uir_shader *s, + struct tgsi_xlat_ctx *ctx, + const struct tgsi_insn *in); + +enum tgsi_xlat_status tgsi_insn_to_uir(struct uir_shader *s, + const struct tgsi_insn *in); + +/* Translate a whole instruction list. On failure *at is the index that failed. */ +enum tgsi_xlat_status tgsi_to_uir(struct uir_shader *s, + const struct tgsi_insn *insns, unsigned n, + unsigned *at); + +const char *tgsi_xlat_status_name(enum tgsi_xlat_status st); + +/* Exposed for the test: is this TGSI opcode translatable at all? */ +int tgsi_opcode_supported(unsigned opcode); + +/* The mapping table itself, so a test can check the numbers this file actually + * uses against Mesa's rather than against a second copy of them. A parallel + * list in the test was the first attempt and it verified nothing: renumbering + * an opcode here left the test passing. */ +struct tgsi_uir_map { + const char *name; /* the TGSI mnemonic, for looking Mesa's up */ + unsigned tgsi; /* the number this file believes */ + int uir; /* enum uir_op */ + unsigned nsrc; +}; + +const struct tgsi_uir_map *tgsi_uir_mapping(unsigned *n); + +#endif diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/usse-core.c mesa-26.2.2/src/gallium/drivers/sgx/usse-core.c --- mesa-26.2.2.orig/src/gallium/drivers/sgx/usse-core.c 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/usse-core.c 2026-09-08 10:57:36.684077307 +0200 @@ -0,0 +1,1737 @@ +/* + * usse-core.c - USSE (PowerVR SGX Series5 / SGX535) instruction decoder + * + * Decodes 64-bit USSE instructions into the text form used by Imagination's + * own Series5 disassembler, so that output can be diffed directly against + * both the vendor tool and the published SGX540 microkernel disassembly. + * + * Field layout was derived from upstream/usse-docs/USSE_ISA.txt and refined + * by exhaustive single-bit probing of the vendor decoder (see README.md). + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#include +#include +#include +#include +#include "usse.h" + +const char *const usp_opcode_name[USP_OPCODE_COUNT] = { + "INVALID", "MAD", "ADM", "MSA", "FRC", "RCP", "RSQ", "LOG", "EXP", "DP", + "DDP", "DDPC", "MIN", "MAX", "DSX", "DSY", "MOVC", "FMAD16", "EFO", + "PCKUNPCK", "TEST", "AND", "OR", "XOR", "SHL", "ROL", "SHR", "ASR", + "RLP", "TESTMASK", "SOP2", "SOP3", "SOPWM", "IMA8", "IMA16", "IMAE", + "ADIF", "BILIN", "FIRV", "FIRH", "DOT3", "DOT4", "FPMA", "SMP", + "SMPBIAS", "SMPREPLACE", "SMPGRAD", "LD", "ST", "BA", "BR", "LAPC", + "SETL", "SAVL", "NOP", "SMOA", "SMR", "SMLSI", "SMBO", "IMO", "SETFC", + "IDF", "WDF", "SETM", "EMIT", "LIMM", "LOCK", "RELEASE", "LDR", "STR", + "WOP", "PCOEFF", "PTOFF", "ATST8", "DEPTHF" +}; + +const char *const usse_bank_name[9] = { + "r", "o", "pa", "sa", "idx", "c", "#", "i", "ireg" +}; + +/* ------------------------------------------------------------------ */ +/* string builder */ + +struct sb { + char *p; + size_t left; + int trunc; +}; + +static void sb_init(struct sb *s, char *buf, size_t len) +{ + s->p = buf; s->left = len; s->trunc = 0; + if (len) buf[0] = 0; +} + +static void sbs(struct sb *s, const char *str) +{ + size_t n = strlen(str); + if (n + 1 > s->left) { s->trunc = 1; n = s->left ? s->left - 1 : 0; } + memcpy(s->p, str, n); + s->p += n; s->left -= n; + if (s->left) *s->p = 0; +} + +static void sbf(struct sb *s, const char *fmt, ...) + __attribute__((format(printf, 2, 3))); + +static void sbf(struct sb *s, const char *fmt, ...) +{ + char tmp[256]; + va_list ap; + va_start(ap, fmt); + vsnprintf(tmp, sizeof tmp, fmt, ap); + va_end(ap); + sbs(s, tmp); +} + +/* ------------------------------------------------------------------ */ +/* common field extraction */ + +/* Predicate, 3-bit form at w1[26:24] (most instructions). */ +static const char *const pred3[8] = { + "", "p0 ", "p1 ", "p2 ", "p3 ", "!p0 ", "!p1 ", "Pn " +}; +/* Predicate, 2-bit form at w1[26:25] (IMA16/IMAE/IMA8/FPMA/SOP...). */ +static const char *const pred2[4] = { "", "p0 ", "p1 ", "!p0 " }; + +/* Source-operand bank, 3-bit {ext, w0 pair}. */ +static const char *const sbank[8] = { "r", "o", "pa", "sa", NULL, "c", NULL, "i" }; +/* Destination bank, 3-bit {w1[19], w1[1:0]}. */ +static const char *const dbank[8] = { "r", "o", "pa", NULL, "sa", "c", NULL, "i" }; + +struct operand { + int bank; /* 0..7 in the respective ordering */ + unsigned num; + int neg, abs; + const char *fmtsuf; /* extra suffix such as ".low" */ +}; + +/* An indexed operand packs {bank, index-register, offset} into the + * register-number field. In F16 mode every subfield shifts down one bit + * because bit 0 then selects the 16-bit half. */ +static void emit_indexed(struct sb *s, unsigned num, int f16) +{ + static const char *const bk[4] = { "r", "o", "pa", "sa" }; + if (f16) + sbf(s, "%s[i%c + #%u]", bk[BF(num, 5, 4)], + BIT(num, 3) ? 'h' : 'l', BF(num, 2, 0)); + else + sbf(s, "%s[i%c + #%u]", bk[BF(num, 6, 5)], + BIT(num, 4) ? 'h' : 'l', BF(num, 3, 0)); +} + +/* f16: 0 = plain 32-bit operand, 1 = the instruction is in F16 mode, where + * bit 6 of the number marks the operand as a 16-bit half-register. + * 2 = as 1 but the vendor omits the component index (DP group). */ +static const char *src_tail; /* format suffix printed after .neg/.abs */ +static int dst_noalias; /* SMP names r124..r127, it does not alias them */ + +static void emit_src_f(struct sb *s, int bank, unsigned num, int neg, int ab, + const char *suf, int f16) +{ + int half = f16 && BIT(num, 6) && bank != 6 && bank != 5; + unsigned n = half && bank != 4 ? BF(num, 5, 1) : num; + int alias = 0; + + if (bank == 0) { /* temps at the bank top alias i0..i3 */ + if (half) { + if (num >= 124) { alias = 1; n = (num - 124) >> 1; } + } else if (f16) { + if (num >= 60 && num < 64) { alias = 1; n = num - 60; } + } else if (num >= 124) { + alias = 1; n = num - 124; + } + } + if (alias) + sbf(s, "i%u", n); + else if (bank == 4) /* indexed */ + emit_indexed(s, num, f16); + else if (bank == 6) /* immediate */ + sbf(s, "#%u", n); + else if (bank == 5) /* special bank: constants, then globals */ + sbf(s, n < 64 ? "c%u" : "g%u", n < 64 ? n : n - 64); + else + sbf(s, "%s%u", sbank[bank], n); + if (suf) sbs(s, suf); + if (neg) sbs(s, ".neg"); + if (ab) sbs(s, ".abs"); + if (src_tail) { sbs(s, src_tail); src_tail = NULL; } + if (half) { + unsigned comp = bank == 4 ? 0 : BIT(num, 0) * 2; + if (f16 == 2 && !comp) sbs(s, ".flt16"); + else sbf(s, ".flt16.%u", comp); + } +} + +static void emit_src(struct sb *s, int bank, unsigned num, int neg, int ab, + const char *suf) +{ + emit_src_f(s, bank, num, neg, ab, suf, 0); +} + +static void emit_dst(struct sb *s, int bank, unsigned num) +{ + if (bank == 3) + emit_indexed(s, num, 0); + else if (bank == 6) + sbf(s, "i.%s%s", BIT(num, 0) ? "l" : "", BIT(num, 1) ? "h" : ""); + else if (bank == 5) + sbf(s, num < 64 ? "c%u" : "g%u", num < 64 ? num : num - 64); + else if (!bank && num >= 124 && !dst_noalias) + sbf(s, "i%u", num - 124); + else + sbf(s, "%s%u", dbank[bank], num); +} + +/* destination write-mask suffix; mask 1 renders as nothing, 0 as a bare dot */ +static void emit_mask(struct sb *s, unsigned mask) +{ + static const char ch[4] = { 'x', 'y', 'z', 'w' }; + char t[8]; + int n = 0, i; + if (mask == 1) return; + t[n++] = '.'; + for (i = 0; i < 4; i++) + if (mask & (1u << i)) t[n++] = ch[i]; + t[n] = 0; + sbs(s, t); +} + +/* repeat / write-mask handling shared by the "standard" ALU encodings: + * w1[21] selects repeat mode, w1[15:12] is then a count, else a mask. */ +static void emit_repeat(struct sb *s, uint32_t w1) +{ + if (BIT(w1, 21)) { + unsigned n = BF(w1, 15, 12) + 1; + if (n > 1) sbf(s, ".repeat%u", n); + } +} + +static void emit_dstmask(struct sb *s, uint32_t w1) +{ + if (!BIT(w1, 21)) emit_mask(s, BF(w1, 15, 12)); +} + +/* skipinv / syncs / nosched / end, in the order the vendor prints them */ +static void emit_stdmods(struct sb *s, uint32_t w1, int has_nosched) +{ + if (BIT(w1, 23)) sbs(s, ".skipinv"); + if (BIT(w1, 20)) sbs(s, ".syncs"); + if (has_nosched && BIT(w1, 11)) sbs(s, ".nosched"); + if (BIT(w1, 18)) sbs(s, ".end"); +} + +/* ------------------------------------------------------------------ */ + +static int src_bank(uint32_t w0, uint32_t w1, int which) +{ + switch (which) { + case 1: return (BIT(w1, 17) << 2) | BF(w0, 31, 30); + case 2: return (BIT(w1, 16) << 2) | BF(w0, 29, 28); + } + return 0; +} + +/* An immediate source-2 borrows the (otherwise unused) source-0 field for + * its upper seven bits, giving a 14-bit constant. */ +static unsigned src2_imm(uint32_t w0) +{ + return BF(w0, 6, 0) | (BF(w0, 20, 14) << 7); +} + +/* Source 0 carries only one bank bit; temp numbers 124..127 alias the four + * internal registers. */ +static void emit_src0_f(struct sb *s, uint32_t w0, uint32_t w1, int neg, int ab, + const char *suf, int f16) +{ + emit_src_f(s, BIT(w1, 2) ? 2 : 0, BF(w0, 20, 14), neg, ab, suf, f16); +} + +/* The MOE offsets are 7-bit signed, but the vendor lets the wider field + * through unmasked when the sign bit is clear. */ +static int moe_s7(unsigned v) +{ + return BIT(v, 6) ? (int)BF(v, 5, 0) - 64 : (int)v; +} + +static int s10(unsigned v) +{ + return BIT(v, 9) ? (int)v - 1024 : (int)v; +} + +static const char swz[4] = { 'x', 'y', 'z', 'w' }; + +/* The extended bank order {r, pa, o, sa}, shared by the source-0 slot, the + * destinations ATST8/DEPTHF/PCOEFF put there and the load/store base. */ +static void emit_ext(struct sb *s, int bank, unsigned num) +{ + static const int bb[4] = { 0, 2, 1, 4 }; + emit_dst(s, bb[bank], num); +} + +/* The opcodes flagged CanUseExtSrc0Banks order the source-0 bank differently */ +static void emit_src0_ext(struct sb *s, uint32_t w0, uint32_t w1) +{ + emit_ext(s, (BIT(w1, 19) << 1) | BIT(w1, 2), BF(w0, 20, 14)); +} + +static void emit_src0(struct sb *s, uint32_t w0, uint32_t w1, int neg, int ab, + const char *suf) +{ + emit_src0_f(s, w0, w1, neg, ab, suf, 0); +} + +/* The integer/C10 groups - SOP2, the 0x11 group and SOPWM - share one operand + * form: a 7-bit field whose top bit is a .c10 format flag for the register + * banks, but the number's high bit for the constant and immediate banks. + * Registers 60..63 name the four internal registers and the indexed banks use + * the narrow layout, the same way the F16 operands do. */ +static void emit_src_c10(struct sb *s, int bank, unsigned field) +{ + unsigned num = field & 0x3f; + + if (bank == 6) { sbf(s, "#%u", field); return; } + if (bank == 5) { sbf(s, field < 64 ? "c%u" : "g%u", num); return; } + if (bank == 4) emit_indexed(s, num, 1); + else if (!bank && num >= 60) sbf(s, "i%u", num - 60); + else sbf(s, "%s%u", sbank[bank], num); + if (BIT(field, 6)) sbs(s, ".c10"); +} + +static void emit_dst_c10(struct sb *s, int bank, unsigned field) +{ + unsigned num = field & 0x3f; + + if (bank == 6) { + sbf(s, "i.%s%s", BIT(num, 0) ? "l" : "", BIT(num, 1) ? "h" : ""); + return; + } + if (bank == 5) { sbf(s, field < 64 ? "c%u" : "g%u", num); return; } + if (bank == 3) emit_indexed(s, num, 1); + else if (!bank && num >= 60) sbf(s, "i%u", num - 60); + else sbf(s, "%s%u", dbank[bank], num); + if (BIT(field, 6)) sbs(s, ".c10"); +} + +/* SOP2 carries a repeat count in w1[14:12]; SOPWM spends those bits on its byte + * mask and the 0x11 group has no repeat at all. */ +static void emit_sop_mods(struct sb *s, uint32_t w1, int rep) +{ + if (BIT(w1, 23)) sbs(s, ".skipinv"); + if (BIT(w1, 22)) sbs(s, ".nosched"); + if (BIT(w1, 18)) sbs(s, ".end"); + if (rep && BF(w1, 14, 12)) sbf(s, ".repeat%u", BF(w1, 14, 12) + 1); +} + +/* ------------------------------------------------------------------ */ +/* opcode classification */ + +enum usp_opcode usse_classify(const struct usse_insn *in) +{ + uint32_t w1 = in->w1, w0 = in->w0; + unsigned op = BF(w1, 31, 27); + + switch (op) { + case 0x00: return (enum usp_opcode)(USP_MAD + BF(w1, 10, 9)); + case 0x01: return (enum usp_opcode)(USP_RCP + BF(w1, 10, 9)); + case 0x02: + if (BF(w1, 10, 9) == 3) return USP_INVALID; + return (enum usp_opcode)(USP_DP + BF(w1, 10, 9)); + case 0x03: + if (BF(w1, 10, 9) >= 2) return USP_INVALID; + return (enum usp_opcode)(USP_MIN + BF(w1, 10, 9)); + case 0x04: + if (BF(w1, 10, 9) >= 2) return USP_INVALID; + return (enum usp_opcode)(USP_DSX + BF(w1, 10, 9)); + case 0x05: return USP_MOVC; + case 0x06: return BF(w1, 10, 9) == 0 ? USP_FMAD16 : USP_INVALID; + case 0x07: return USP_EFO; + case 0x08: return USP_PCKUNPCK; + case 0x09: return USP_TEST; + case 0x0A: return BIT(w1, 3) ? USP_OR : USP_AND; + case 0x0B: return USP_XOR; + case 0x0C: return BIT(w1, 3) ? USP_ROL : USP_SHL; + case 0x0D: return BIT(w1, 3) ? USP_ASR : USP_SHR; + case 0x0E: return USP_RLP; + case 0x0F: return USP_TESTMASK; + case 0x10: return USP_SOP2; + case 0x11: return USP_SOP3; + case 0x12: return USP_SOPWM; + case 0x13: return USP_IMA8; + case 0x14: return USP_IMA16; + case 0x15: return USP_IMAE; + case 0x16: + switch (BF(w1, 21, 20)) { + case 0: return USP_ADIF; + case 2: return USP_BILIN; + case 3: return USP_FIRV; + } + return USP_INVALID; + case 0x17: return USP_FIRH; + case 0x18: return BIT(w1, 24) ? USP_DOT4 : USP_DOT3; + case 0x19: return USP_FPMA; + case 0x1A: case 0x1B: return USP_INVALID; + case 0x1C: return (enum usp_opcode)(USP_SMP + BF(w1, 9, 8)); + case 0x1D: return USP_LD; + case 0x1E: return USP_ST; + case 0x1F: + switch (BF(w1, 21, 20)) { + case 0: + switch (BF(w1, 8, 6)) { + case 0: return USP_BA; + case 1: return USP_BR; + case 2: return USP_LAPC; + case 3: return USP_SETL; + case 4: return USP_SAVL; + case 5: return USP_NOP; + } + return USP_INVALID; + case 1: + switch (BF(w1, 26, 24)) { + case 0: return USP_SMOA; + case 1: return USP_SMR; + case 2: return USP_SMLSI; + case 3: return USP_SMBO; + case 4: return USP_IMO; + case 5: return USP_SETFC; + } + return USP_INVALID; + case 2: + switch (BF(w1, 26, 24)) { + case 0: return USP_IDF; + case 1: return USP_WDF; + case 2: return USP_SETM; + case 3: return USP_EMIT; + case 4: return USP_LIMM; + case 5: return BIT(w0, 0) ? USP_RELEASE : USP_LOCK; + case 6: return BIT(w1, 19) ? USP_STR : USP_LDR; + case 7: return USP_WOP; + } + return USP_INVALID; + case 3: + switch (BF(w1, 26, 24)) { + case 0: return BIT(w1, 15) ? USP_PTOFF : USP_PCOEFF; + case 1: return USP_ATST8; + case 3: return USP_DEPTHF; + } + return USP_INVALID; + } + return USP_INVALID; + } + return USP_INVALID; +} + +/* ------------------------------------------------------------------ */ +/* per-group formatting */ + +/* group 0x00: MAD / ADM / MSA / FRC */ +static void dis_mad(struct sb *s, uint32_t w0, uint32_t w1) +{ + static const char *const nm[4] = { "fmad", "fadm", "fmsa", "fsubflr" }; + unsigned sub = BF(w1, 10, 9); + int f16 = BIT(w1, 22) ? (sub == 3 ? 2 : 1) : 0; + sbs(s, pred3[BF(w1, 26, 24)]); + sbs(s, nm[sub]); + emit_stdmods(s, w1, 1); + emit_repeat(s, w1); + sbs(s, " "); + emit_dst(s, (BIT(w1, 19) << 2) | BF(w1, 1, 0), BF(w0, 27, 21)); + emit_dstmask(s, w1); + sbs(s, ", "); + if (sub != 3) { /* FRC has no source 0 */ + emit_src0_f(s, w0, w1, BIT(w1, 7), BIT(w1, 8), NULL, f16); + sbs(s, ", "); + } + emit_src_f(s, src_bank(w0, w1, 1), BF(w0, 13, 7), + BIT(w1, 5), BIT(w1, 6), NULL, f16); + sbs(s, ", "); + emit_src_f(s, src_bank(w0, w1, 2), BF(w0, 6, 0), + BIT(w1, 3), BIT(w1, 4), NULL, f16); +} + +/* group 0x01: RCP / RSQ / LOG / EXP - single source with a format selector */ +static void dis_rcp(struct sb *s, uint32_t w0, uint32_t w1) +{ + static const char *const nm[4] = { "frcp", "frsq", "flog", "fexp" }; + unsigned f = BF(w1, 8, 7); + char fmt[16]; + if (f == 3) { sbs(s, "invalid_complex_data_type"); return; } + sprintf(fmt, ".%s.%u", f == 1 ? "flt16" : "c10", + f == 1 ? BIT(w1, 3) * 2 : BF(w1, 4, 3)); + sbs(s, pred3[BF(w1, 26, 24)]); + sbs(s, nm[BF(w1, 10, 9)]); + if (BIT(w1, 23)) sbs(s, ".skipinv"); + if (BIT(w1, 20)) sbs(s, ".syncs"); + if (BIT(w1, 11)) sbs(s, ".nosched"); + if (BIT(w1, 18)) sbs(s, ".end"); + if (BIT(w1, 2)) sbs(s, ".tpres"); + emit_repeat(s, w1); + sbs(s, " "); + emit_dst(s, (BIT(w1, 19) << 2) | BF(w1, 1, 0), BF(w0, 27, 21)); + emit_dstmask(s, w1); + sbs(s, ", "); + src_tail = f ? fmt : NULL; + emit_src(s, src_bank(w0, w1, 1), BF(w0, 13, 7), + BIT(w1, 5), BIT(w1, 6), NULL); + src_tail = NULL; +} + +/* groups 0x02..0x04, 0x06: two-source float ops sharing one shape */ +static void dis_2src(struct sb *s, uint32_t w0, uint32_t w1, const char *name, + int nsrc) +{ + int f16 = BIT(w1, 22); + sbs(s, pred3[BF(w1, 26, 24)]); + sbs(s, name); + emit_stdmods(s, w1, 1); + emit_repeat(s, w1); + sbs(s, " "); + emit_dst(s, (BIT(w1, 19) << 2) | BF(w1, 1, 0), BF(w0, 27, 21)); + emit_dstmask(s, w1); + if (nsrc >= 3) { + sbs(s, ", "); + emit_src0_f(s, w0, w1, BIT(w1, 7), BIT(w1, 8), NULL, f16); + } + sbs(s, ", "); + emit_src_f(s, src_bank(w0, w1, 1), BF(w0, 13, 7), + BIT(w1, 5), BIT(w1, 6), NULL, f16); + sbs(s, ", "); + emit_src_f(s, src_bank(w0, w1, 2), BF(w0, 6, 0), + BIT(w1, 3), BIT(w1, 4), NULL, f16); +} + +/* group 0x07: EFO - the "extended floating point operation" unit. Sources with + * bit 6 of their number set render as an F16 register pair. */ +static void efo_src(struct sb *s, int bank, unsigned num, int neg, int ab) +{ + emit_src_f(s, bank, num, neg, ab, NULL, 0); + if (bank == 0 && num >= 60 && num < 64) { /* pairs with i0..i3 */ + sbf(s, "/i%u", num - 60); + if (neg) sbs(s, ".neg"); + if (ab) sbs(s, ".abs"); + return; + } + if (BIT(num, 6)) { + sbs(s, "/"); + if (bank == 0 && num >= 124) { + sbf(s, "i%u", (num - 124) >> 1); + if (neg) sbs(s, ".neg"); + if (ab) sbs(s, ".abs"); + } + else + emit_src_f(s, bank, BF(num, 5, 1), neg, ab, NULL, 0); + sbf(s, ".flt16.%u", BIT(num, 0) * 2); + } +} + +static void dis_efo(struct sb *s, uint32_t w0, uint32_t w1) +{ + static const char *const dsel[4] = { "i0", "i1", "a0", "a1" }; + static const char *const isel[4][2] = { + { "a0", "a1" }, { "a1", "a0" }, { "m0", "m1" }, { "a0", "m1" } + }; + static const char *const msel[4] = { + "m0=src0*src1, m1=src0*src2", "m0=src0*src1, m1=src0*src0", + "m0=src1*src2, m1=src0*src0", "m0=src1*i0, m1=src0*i1" + }; + unsigned a = BF(w1, 17, 16), i = BF(w1, 19, 18), rp = BF(w1, 13, 12) + 1; + const char *ng = BIT(w1, 22) ? "-" : ""; + + sbs(s, pred3[BF(w1, 26, 24)]); + sbs(s, "efo"); + if (BIT(w1, 23)) sbs(s, ".skipinv"); + if (BIT(w1, 11)) sbs(s, ".nosched"); + if (rp > 1) sbf(s, ".repeat%u", rp); + sbs(s, " "); + if (BF(w1, 1, 0) == 0 && BF(w0, 27, 21) >= 124) + sbf(s, "i%u", BF(w0, 27, 21) - 124); + else + emit_dst(s, BF(w1, 1, 0), BF(w0, 27, 21)); + sbf(s, "= %s, ", dsel[BF(w1, 21, 20)]); + sbf(s, "%si0 = %s, ", BIT(w1, 10) ? "" : "!", isel[i][0]); + sbf(s, "%si1 = %s, ", BIT(w1, 9) ? "" : "!", isel[i][1]); + switch (a) { + case 0: sbf(s, "a0=m0+m1, a1=%si1+i0, ", ng); break; + case 1: sbf(s, "a0=m0+src2, a1=%si1+i0, ", ng); break; + case 2: sbf(s, "a0=i0+m0, a1=%si1+m1, ", ng); break; + default: sbf(s, "a0=src0+src1 a1=%ssrc2+src0, ", ng); break; + } + sbf(s, "%s, ", msel[BF(w1, 15, 14)]); + efo_src(s, BIT(w1, 2) ? 2 : 0, BF(w0, 20, 14), BIT(w1, 7), BIT(w1, 8)); + sbs(s, ", "); + efo_src(s, BF(w0, 31, 30), BF(w0, 13, 7), BIT(w1, 5), BIT(w1, 6)); + sbs(s, ", "); + efo_src(s, BF(w0, 29, 28), BF(w0, 6, 0), BIT(w1, 3), BIT(w1, 4)); +} + +/* group 0x06: FMAD16 - the sources are already 16-bit, so no F16 remap */ +static void dis_fmad16(struct sb *s, uint32_t w0, uint32_t w1) +{ + sbs(s, pred3[BF(w1, 26, 24)]); + sbs(s, "fmad16"); + emit_stdmods(s, w1, 1); + emit_repeat(s, w1); + sbs(s, " "); + emit_dst(s, (BIT(w1, 19) << 2) | BF(w1, 1, 0), BF(w0, 27, 21)); + emit_dstmask(s, w1); + sbs(s, ", "); + emit_src0(s, w0, w1, BIT(w1, 7), BIT(w1, 8), NULL); + sbs(s, ", "); + emit_src(s, src_bank(w0, w1, 1), BF(w0, 13, 7), BIT(w1, 5), BIT(w1, 6), NULL); + sbs(s, ", "); + emit_src(s, src_bank(w0, w1, 2), BF(w0, 6, 0), BIT(w1, 3), BIT(w1, 4), NULL); +} + +/* group 0x02: DP / DDP / DDPC. Bit 14 of word 0 turns the plain dot product + * into its clip-plane form. */ +static void dis_dp(struct sb *s, uint32_t w0, uint32_t w1) +{ + unsigned sub = BF(w1, 10, 9); + int clip = BIT(w0, 14), f16; + if (sub == 3) { sbs(s, "invalid_float_op2"); return; } + f16 = BIT(w1, 22) ? (sub == 0 && !clip ? 1 : 2) : 0; + sbs(s, pred3[BF(w1, 26, 24)]); + sbs(s, sub == 0 ? (clip ? "fdpc" : "fdp") : sub == 1 ? "fddp" : "fddpc"); + emit_stdmods(s, w1, 1); + emit_repeat(s, w1); + sbs(s, " "); + if (sub == 0) + emit_dst(s, (BIT(w1, 19) << 2) | BF(w1, 1, 0), BF(w0, 27, 21)); + else + emit_dst(s, BF(w1, 1, 0) == 3 ? 3 : BF(w1, 1, 0), BF(w0, 27, 21)); + emit_dstmask(s, w1); + sbs(s, ", "); + if (sub != 0) { + sbf(s, "i%u, ", BIT(w1, 19)); + if (sub == 2) sbf(s, "cp%u, cp%u, ", BF(w1, 5, 3), BF(w1, 8, 6)); + if (sub == 2) { + emit_src0_f(s, w0, w1, 0, 0, NULL, f16); + sbs(s, ", "); + emit_src_f(s, src_bank(w0, w1, 1), BF(w0, 13, 7), + 0, 0, NULL, f16); + sbs(s, ", "); + emit_src_f(s, src_bank(w0, w1, 2), BF(w0, 6, 0), + 0, 0, NULL, f16); + return; + } + emit_src0_f(s, w0, w1, BIT(w1, 7), BIT(w1, 8), NULL, f16); + sbs(s, ", "); + emit_src_f(s, src_bank(w0, w1, 1), BF(w0, 13, 7), + BIT(w1, 5), BIT(w1, 6), NULL, f16); + sbs(s, ", "); + emit_src_f(s, src_bank(w0, w1, 2), BF(w0, 6, 0), + BIT(w1, 3), BIT(w1, 4), NULL, f16); + return; + } + if (clip) sbf(s, "cp%u, ", BF(w0, 17, 15)); + emit_src_f(s, src_bank(w0, w1, 1), BF(w0, 13, 7), + BIT(w1, 5), BIT(w1, 6), NULL, f16); + sbs(s, ", "); + emit_src_f(s, src_bank(w0, w1, 2), BF(w0, 6, 0), + BIT(w1, 3), BIT(w1, 4), NULL, f16); +} + +/* group 0x05: MOVC - plain move, or a conditional move with a test type */ +static void dis_movc(struct sb *s, uint32_t w0, uint32_t w1) +{ + static const char *const dt[8] = { + NULL, ".i8", ".i16", ".i32", ".flt", ".i10", NULL, NULL + }; + static const char *const tst[4] = { ".tz", ".tnz", ".tn", ".tp|z" }; + unsigned t = BF(w1, 10, 8); + if (t >= 6) { sbs(s, "invalid_movc_data_type"); return; } + sbs(s, pred3[BF(w1, 26, 24)]); + sbs(s, t ? "movc" : "mov"); + if (BIT(w1, 23)) sbs(s, ".skipinv"); + if (BIT(w1, 20)) sbs(s, ".syncs"); + if (BIT(w1, 11)) sbs(s, ".nosched"); + if (BIT(w1, 18)) sbs(s, ".end"); + if (t == 1) { /* the i8 case emits repeat first */ + emit_repeat(s, w1); + sbs(s, dt[t]); + } else if (t) { + sbs(s, dt[t]); + emit_repeat(s, w1); + } else { + emit_repeat(s, w1); + } + if (t) sbs(s, tst[(BIT(w1, 22) << 1) | BIT(w1, 7)]); + sbs(s, " "); + emit_dst(s, (BIT(w1, 19) << 2) | BF(w1, 1, 0), BF(w0, 27, 21)); + emit_dstmask(s, w1); + sbs(s, ", "); + if (t) { + emit_src0(s, w0, w1, 0, 0, NULL); + sbs(s, ", "); + } + if (!t && src_bank(w0, w1, 1) == 6) /* a plain mov prints it in hex */ + sbf(s, "#0x%08X", BF(w0, 13, 7)); + else + emit_src(s, src_bank(w0, w1, 1), BF(w0, 13, 7), 0, 0, NULL); + if (t) { + int b = src_bank(w0, w1, 2); + sbs(s, ", "); + emit_src(s, b, BF(w0, 6, 0), 0, 0, NULL); + } +} + +/* group 0x08: PCKUNPCK. word1[11:9] selects the source format and [8:6] the + * destination format; the vendor spells the pair out in the mnemonic. */ +const char *const pck_name[64] = { + "unpcku8u8", "unpcku8s8", NULL, "unpcku16u8", "unpcks16u8", + "unpckf16u8", "unpckf32u8", NULL, + "statomic", NULL, NULL, "unpcku16s8", "unpcks16s8", "unpckf16s8", + "unpckf32s8", NULL, + NULL, NULL, NULL, "unpcku16o8", "unpcks16o8", "unpckf16o8", + "unpckf32o8", NULL, + "pcku8u16", "pcks8u16", "pcko8u16", "pcku16u16", "pcks16u16", + "pckf16u16", "unpckf32u16", "pckc10u16", + "pcku8s16", "pcks8s16", "pcko8s16", "pcku16s16", "pcks16s16", + "pckf16s16", "unpckf32s16", "pckc10s16", + "pcku8f16", "pcks8f16", "pcko8f16", "pcku16f16", "pcks16f16", + "pckf16f16", "unpckf32f16", "pckc10f16", + "pcku8f32", "pcks8f32", "pcko8f32", "pcku16f32", "pcks16f32", + "pckf16f32", "unpckf32f32", "pckc10f32", + NULL, NULL, NULL, "unpcku16c10", "unpcks16c10", "unpckf16c10", + "unpckf32c10", "unpckc10c10" +}; + +static void dis_pck(struct sb *s, uint32_t w0, uint32_t w1) +{ + unsigned f = BF(w1, 11, 6), bm = BF(w1, 5, 2); + int atomic; + if (!pck_name[f]) { sbs(s, "invalid_pck_format_combination"); return; } + atomic = !strcmp(pck_name[f], "statomic"); + sbs(s, pred3[BF(w1, 26, 24)]); + sbs(s, pck_name[f]); + if (BIT(w1, 23)) sbs(s, ".skipinv"); + if (BIT(w1, 20)) sbs(s, ".syncs"); + if (BIT(w1, 22)) sbs(s, ".nosched"); + if (BIT(w1, 18)) sbs(s, ".end"); + if (BIT(w0, 18)) sbs(s, ".scale"); + emit_repeat(s, w1); + sbs(s, " "); + emit_dst(s, (BIT(w1, 19) << 2) | BF(w1, 1, 0), BF(w0, 27, 21)); + if (bm != 15) + sbf(s, ".bytemask%u%u%u%u", BIT(bm, 3), BIT(bm, 2), + BIT(bm, 1), BIT(bm, 0)); + if (!BIT(w1, 21)) emit_mask(s, BF(w1, 15, 12)); + sbs(s, ", "); + emit_src(s, src_bank(w0, w1, 1), BF(w0, 13, 7), 0, 0, NULL); + if (!atomic || BF(w0, 17, 16)) sbf(s, ".%u", BF(w0, 17, 16)); + sbs(s, ", "); + emit_src(s, src_bank(w0, w1, 2), BF(w0, 6, 0), 0, 0, NULL); + if (!atomic || BF(w0, 15, 14)) sbf(s, ".%u", BF(w0, 15, 14)); + sbs(s, ", nearest"); +} + +/* group 0x10: SOP2 - two sum-of-products units, colour and alpha independent. + * Field positions mirror SOPWM, but the operand select table differs and the + * two ops live in w0 rather than w1. Mapped by sweeping all 64 bits through + * Imagination's decoder, and both select fields swept over all their values + * from three independent base words. */ +static void dis_sop2(struct sb *s, uint32_t w0, uint32_t w1) +{ + /* The two operands do not share one select table: they agree up to the + * alpha-saturate slot and then diverge, source 1 taking s2scale where + * source 2 takes zeros. Same split on the alpha side. */ + static const char *const csel1[8] = { + "zero", "s1", "s2", "s1a", "s2a", "asat", "s2scale", NULL + }; + static const char *const csel2[8] = { + "zero", "s1", "s2", "s1a", "s2a", "asat", "zeros", NULL + }; + static const char *const asel1[4] = { "zero", "s1a", "s2a", "s2scale" }; + static const char *const asel2[4] = { "zero", "s1a", "s2a", "zeros" }; + static const char *const op[4] = { "add", "sub", "min", "max" }; + unsigned c1 = BF(w1, 8, 6), c2 = BF(w1, 5, 3); + unsigned a1 = BF(w1, 21, 20), a2 = BF(w1, 10, 9); + + if (!csel1[c1]) { sbs(s, "invalid_sop2_csel1"); return; } + if (!csel2[c2]) { sbs(s, "invalid_sop2_csel2"); return; } + sbs(s, pred2[BF(w1, 26, 25)]); + sbs(s, "sop2"); + emit_sop_mods(s, w1, 1); + sbs(s, " "); + emit_dst_c10(s, (BIT(w1, 19) << 2) | BF(w1, 1, 0), BF(w0, 27, 21)); + sbs(s, ", "); + emit_src_c10(s, src_bank(w0, w1, 1), BF(w0, 13, 7)); + sbs(s, ", "); + emit_src_c10(s, src_bank(w0, w1, 2), BF(w0, 6, 0)); + /* each side may carry a bare "comp" operand ahead of its selects */ + if (BIT(w0, 20)) sbs(s, ", comp"); + sbf(s, ", %s%s", csel1[c1], BIT(w1, 24) ? ".comp" : ""); + sbf(s, ", %s%s", csel2[c2], BIT(w1, 15) ? ".comp" : ""); + sbf(s, ", %s+asop2 ", op[BF(w0, 19, 18)]); + if (BIT(w0, 15)) sbs(s, "comp, "); + sbf(s, "%s%s", asel1[a1], BIT(w1, 11) ? ".comp" : ""); + sbf(s, ", %s%s", asel2[a2], BIT(w1, 2) ? ".comp" : ""); + sbf(s, ", %s", op[BF(w0, 17, 16)]); + if (BIT(w0, 14)) sbs(s, ", neg"); +} + +/* group 0x11: a three-source group with three sub-forms. w1[21] picks SOP3 + * versus the LRPs and w1[10] picks LRP1 from LRP2. Field meanings differ per + * sub-form, which is why single-bit flips from one base looked incoherent - + * each field had to be enumerated across all its values. + * + * All three share the register layout. Source 0 is the odd one: it has a single + * bank bit, w1[2], choosing r or pa, and reaches the internal registers through + * the 60..63 aliases rather than through a bank code. */ +static const char *const sop3_sel[8] = { + "zero", "asat", "s1", "s1a", "s0", "s0a", "s2", "s2a" +}; +static const char *const sop3_asel[4] = { "zero", "s0a", "s1a", "s2a" }; + +static void emit_sop3_regs(struct sb *s, uint32_t w0, uint32_t w1) +{ + emit_dst_c10(s, (BIT(w1, 19) << 2) | BF(w1, 1, 0), BF(w0, 27, 21)); + sbs(s, ", "); + emit_src_c10(s, BIT(w1, 2) ? 2 : 0, BF(w0, 20, 14)); + sbs(s, ", "); + emit_src_c10(s, src_bank(w0, w1, 1), BF(w0, 13, 7)); + sbs(s, ", "); + emit_src_c10(s, src_bank(w0, w1, 2), BF(w0, 6, 0)); +} + +/* SOP3's alpha side has four forms in w1[11:10]: asop, alrp, arsop, and alrp + * again with a negate on the colour operation. Each prints a different operand + * count, and asop's second factor is the colour select 2 reprinted - the same + * field, complement and all. */ +static void dis_sop3(struct sb *s, uint32_t w0, uint32_t w1) +{ + unsigned form = BF(w1, 11, 10); + + sbs(s, pred2[BF(w1, 26, 25)]); + sbs(s, "sop3"); + emit_sop_mods(s, w1, 0); + sbs(s, " "); + emit_sop3_regs(s, w0, w1); + sbf(s, ", %s%s", sop3_sel[BF(w1, 8, 6)], BIT(w1, 24) ? ".comp" : ""); + sbf(s, ", %s%s", sop3_sel[BF(w1, 5, 3)], BIT(w1, 15) ? ".comp" : ""); + sbf(s, ", %s", BIT(w1, 20) ? "sub" : "add"); + if (form == 3) sbs(s, ", neg"); + if (form == 1 || form == 3) { + sbf(s, "+alrp %s%s", BIT(w1, 13) ? "s1a" : "zero", + BIT(w1, 9) ? ".comp" : ""); + sbf(s, ", %s%s", BIT(w1, 12) ? "s2a" : "zero", + BIT(w1, 14) ? ".comp" : ""); + return; + } + sbf(s, "+%s %s%s", form ? "arsop" : "asop", sop3_asel[BF(w1, 13, 12)], + BIT(w1, 14) ? ".comp" : ""); + if (!form) + sbf(s, ", %s%s", sop3_sel[BF(w1, 5, 3)], + BIT(w1, 15) ? ".comp" : ""); + sbf(s, ", %s", BIT(w1, 9) ? "sub" : "add"); +} + +static void dis_lrp1(struct sb *s, uint32_t w0, uint32_t w1) +{ + sbs(s, pred2[BF(w1, 26, 25)]); + sbs(s, "lrp1"); + emit_sop_mods(s, w1, 0); + sbs(s, " "); + emit_sop3_regs(s, w0, w1); + /* two one-bit selects with their own complements */ + sbf(s, ", %s%s", BIT(w1, 13) ? "s1" : "zero", BIT(w1, 20) ? ".comp" : ""); + sbf(s, ", %s%s", BIT(w1, 12) ? "s2" : "zero", BIT(w1, 14) ? ".comp" : ""); + sbf(s, ", %s+asop ", BIT(w1, 11) ? "s0a" : "s0"); + sbf(s, "%s%s", sop3_sel[BF(w1, 8, 6)], BIT(w1, 24) ? ".comp" : ""); + sbf(s, ", %s%s", sop3_sel[BF(w1, 5, 3)], BIT(w1, 15) ? ".comp" : ""); + sbf(s, ", %s", BIT(w1, 9) ? "sub" : "add"); +} + +/* LRP2 spends the colour side on four selects and keeps SOP3's alpha lerp. Its + * third and fourth operands are one 2-bit field, not two independent bits: the + * code either substitutes "one" for the third or complements one of the pair. */ +static const char *const lrp2_cd[4][2] = { + { "s1", "s2" }, { "s1", "s2.comp" }, { "one", "s2" }, { "s1.comp", "s2" } +}; + +static void dis_lrp2(struct sb *s, uint32_t w0, uint32_t w1) +{ + unsigned cd = (BIT(w1, 20) << 1) | BIT(w1, 11); + + sbs(s, pred2[BF(w1, 26, 25)]); + sbs(s, "lrp2"); + emit_sop_mods(s, w1, 0); + sbs(s, " "); + emit_sop3_regs(s, w0, w1); + sbf(s, ", %s%s", sop3_sel[BF(w1, 8, 6)], BIT(w1, 24) ? ".comp" : ""); + sbf(s, ", %s%s", sop3_sel[BF(w1, 5, 3)], BIT(w1, 15) ? ".comp" : ""); + sbf(s, ", %s, %s+alrp ", lrp2_cd[cd][0], lrp2_cd[cd][1]); + sbf(s, "%s%s", BIT(w1, 13) ? "s1a" : "zero", BIT(w1, 9) ? ".comp" : ""); + sbf(s, ", %s%s", BIT(w1, 12) ? "s2a" : "zero", BIT(w1, 14) ? ".comp" : ""); +} + +/* group 0x12: SOPWM - sum of products with a byte write mask */ +static void dis_sopwm(struct sb *s, uint32_t w0, uint32_t w1) +{ + static const char *const sel[8] = { + "zero", "asat", "s1", "s1a", NULL, NULL, "s2", "s2a" + }; + static const char *const cop[4] = { "add", "sub", "min", "max" }; + unsigned s1 = BF(w1, 8, 6), s2 = BF(w1, 5, 3); + if (!sel[s1]) { sbs(s, "invalid_sopwm_sel1"); return; } + if (!sel[s2]) { sbs(s, "invalid_sopwm_sel2"); return; } + sbs(s, pred2[BF(w1, 26, 25)]); + sbs(s, "sop2wm"); + emit_sop_mods(s, w1, 0); + sbs(s, " "); + emit_dst_c10(s, (BIT(w1, 19) << 2) | BF(w1, 1, 0), BF(w0, 27, 21)); + sbf(s, ".bytemask%u%u%u%u", BIT(w1, 11), BIT(w1, 14), BIT(w1, 13), + BIT(w1, 12)); + sbs(s, ", "); + emit_src_c10(s, src_bank(w0, w1, 1), BF(w0, 13, 7)); + sbs(s, ", "); + emit_src_c10(s, src_bank(w0, w1, 2), BF(w0, 6, 0)); + sbf(s, ", %s%s", sel[s1], BIT(w1, 24) ? ".comp" : ""); + sbf(s, ", %s%s", sel[s2], BIT(w1, 15) ? ".comp" : ""); + sbf(s, ", %s, %s", cop[BF(w1, 21, 20)], cop[BF(w1, 10, 9)]); +} + +/* ALU sub-operation selector shared by TEST and TESTMASK, w0[19:14] */ +static const char *const test_alu[64] = { + "fadd", NULL, NULL, "fsubflr", "frcp", "frsq", "flog", "fexp", + "fdp", "fmin", "fmax", "fdsx", "fdsy", "fmul", "fsub", NULL, + NULL, NULL, NULL, NULL, NULL, NULL, "iadd16", "isub16", + "imul16", "iaddu16", "isubu16", "imulu16", "iadd32", "iaddu32", NULL, NULL, + "iadd8", "isub8", "iaddu8", "isubu8", "imul8", "fpmul8", "imulu8", "fpadd8", + "fpsub8", NULL, NULL, NULL, NULL, NULL, NULL, NULL, + "and", "or", "xor", "shl", "shr", "rol", NULL, "asr", + NULL, NULL, NULL, NULL, NULL, NULL, NULL, NULL +}; +/* number of sources each ALU sub-operation prints */ +static int test_alu_nsrc(unsigned a) +{ + return (a >= 4 && a <= 7) ? 1 : 2; +} + +/* build the ".test" suffix from w1[11:7] */ +static int test_suffix(char *out, unsigned tt) +{ + static const char sign[4] = { 't', 'n', 'p', '?' }; + static const char *const z[8] = { + "ot", "at", "oz", "az", "onz", "anz", NULL, NULL + }; + unsigned sg = tt >> 3, zz = tt & 7; + if (sg == 3) return -2; + if (!z[zz]) return -1; + sprintf(out, ".test%c%s", sign[sg], z[zz]); + return 0; +} + +static const char *const test_chan[8] = { + "chan0", "chan1", "chan2", "chan3", "andall", "orall", "and02", "or02" +}; + +/* w1[18] is overloaded by ALU class: F16 operands on the float ops, C10 on the + * three fp8 ones, and the .partial flag on the bitwise ops. */ +static int test_fmt(unsigned a) +{ + if (a <= 14) return 1; + if (a == 37 || a == 39 || a == 40) return 2; + return 0; +} + +/* fadd, fsubflr, fmul and fsub drop a zero F16 component index */ +static int test_drops_comp(unsigned a) +{ + return a == 0 || a == 3 || a == 13 || a == 14; +} + +/* only the integer 8-bit ALU ops carry the w1[22] saturate flag */ +static int test_has_sat(unsigned a) +{ + return (a >= 32 && a <= 36) || a == 38; +} + +/* group 0x09: TEST */ +static void dis_test(struct sb *s, uint32_t w0, uint32_t w1) +{ + unsigned a = BF(w0, 19, 14); + int fmt, f16, ext = BIT(w1, 18); + char suf[16]; + int r; + if (!test_alu[a]) { sbs(s, "invalid_test_aluop"); return; } + r = test_suffix(suf, BF(w1, 11, 7)); + if (r == -1) { sbs(s, "invalid_test_zero_test"); return; } + if (r == -2) { sbs(s, "invalid_test_sign_test"); return; } + fmt = ext ? test_fmt(a) : 0; + f16 = fmt != 1 ? 0 : test_drops_comp(a) ? 2 : 1; + sbs(s, pred3[BF(w1, 26, 24)]); + sbs(s, test_alu[a]); + if (BIT(w1, 23)) sbs(s, ".skipinv"); + if (BIT(w1, 20)) sbs(s, ".syncs"); + if (ext && a >= 48) sbs(s, ".partial"); + if (BIT(w1, 21)) sbs(s, ".onceonly"); + if (BIT(w1, 22) && test_has_sat(a)) sbs(s, ".sat"); + sbs(s, suf); + sbf(s, ".%s ", test_chan[BF(w1, 6, 4)]); + if (!BIT(w0, 20)) sbs(s, "!"); + if (fmt == 2) + emit_dst_c10(s, (BIT(w1, 19) << 2) | BF(w1, 1, 0), BF(w0, 27, 21)); + else + emit_dst(s, (BIT(w1, 19) << 2) | BF(w1, 1, 0), BF(w0, 27, 21)); + emit_mask(s, BF(w1, 15, 12)); + sbf(s, ", p%u, ", BF(w1, 3, 2)); + if (a >= 48 && src_bank(w0, w1, 1) == 6) + sbf(s, "#0x%08X", BF(w0, 13, 7)); + else if (fmt == 2) + emit_src_c10(s, src_bank(w0, w1, 1), BF(w0, 13, 7)); + else + emit_src_f(s, src_bank(w0, w1, 1), BF(w0, 13, 7), 0, 0, NULL, f16); + if (test_alu_nsrc(a) == 2) { + int b = src_bank(w0, w1, 2); + sbs(s, ", "); + if (fmt == 2) + emit_src_c10(s, b, BF(w0, 6, 0)); + else + emit_src_f(s, b, BF(w0, 6, 0), 0, 0, NULL, f16); + } +} + +/* group 0x0F: TESTMASK - like TEST but writes a mask, no predicate operand */ +static void dis_testmask(struct sb *s, uint32_t w0, uint32_t w1) +{ + static const char *const mk[4] = { "bmsk", "wmsk", "dmsk", NULL }; + unsigned a = BF(w0, 19, 14), mt = BF(w1, 5, 4); + int fmt, f16, ext = BIT(w1, 18); + char suf[16]; + int r; + if (!test_alu[a]) { sbs(s, "invalid_test_aluop"); return; } + r = test_suffix(suf, BF(w1, 11, 7)); + if (r == -1) { sbs(s, "invalid_test_zero_test"); return; } + if (r == -2) { sbs(s, "invalid_test_sign_test"); return; } + if (!mk[mt]) { sbs(s, "invalid_test_mask_type"); return; } + fmt = ext ? test_fmt(a) : 0; + f16 = fmt != 1 ? 0 : test_drops_comp(a) ? 2 : 1; + sbs(s, pred3[BF(w1, 26, 24)]); + sbs(s, test_alu[a]); + if (BIT(w1, 23)) sbs(s, ".skipinv"); + if (BIT(w1, 20)) sbs(s, ".syncs"); + if (ext && a >= 48) sbs(s, ".partial"); + if (BIT(w1, 21)) sbs(s, ".onceonly"); + if (BIT(w1, 22) && test_has_sat(a)) sbs(s, ".sat"); + sbs(s, suf); + sbf(s, ".%s ", mk[mt]); + if (!BIT(w0, 20)) sbs(s, "!"); + if (fmt == 2) + emit_dst_c10(s, (BIT(w1, 19) << 2) | BF(w1, 1, 0), BF(w0, 27, 21)); + else + emit_dst(s, (BIT(w1, 19) << 2) | BF(w1, 1, 0), BF(w0, 27, 21)); + emit_mask(s, BF(w1, 15, 12)); + sbs(s, ", "); + if (fmt == 2) + emit_src_c10(s, src_bank(w0, w1, 1), BF(w0, 13, 7)); + else + emit_src_f(s, src_bank(w0, w1, 1), BF(w0, 13, 7), 0, 0, NULL, f16); + if (test_alu_nsrc(a) == 2) { + int b = src_bank(w0, w1, 2); + sbs(s, ", "); + if (a >= 48 && b == 6) /* bitwise: a bare 32-bit constant */ + sbf(s, "#0x%08X", BF(w0, 6, 0)); + else if (fmt == 2) + emit_src_c10(s, b, BF(w0, 6, 0)); + else + emit_src_f(s, b, BF(w0, 6, 0), 0, 0, NULL, f16); + } +} + +/* groups 0x0A..0x0D: bitwise ops. An immediate source 2 is a 16-bit field + * scaled by the built-in shifter; the vendor decoder folds the shift into + * the printed constant, which is why the shift never appears in the text. */ +/* SMP and its three variants share one encoding: w1[9:8] picks the variant and + * w1[11:10] the coordinate dimensionality, so smp1d..smp3d and the bias, + * replace and grad forms all come out of one case. The operands are + * destination, coordinate, texture state, then the dependent-read counter. + * + * It does not use the operand frame the rest of this group does: there is no + * source 2, w0[31:30] carries the texture-state bank where the others put + * source 1's, and w1[18] is a bank bit rather than .end. Field positions were + * recovered by bit-flip probing Imagination's decoder, see + * work/usse-unmodelled/. */ +static void dis_smp(struct sb *s, uint32_t w0, uint32_t w1) +{ + static const char *const var[4] = { "", "bias", "replace", "grad" }; + static const int s0bank[4] = { 0, 2, 1, 3 }; /* r, pa, o, sa */ + static const char *const s0fmt[4] = { NULL, ".flt16", ".c10", NULL }; + unsigned dim = BF(w1, 11, 10), fmt = BF(w1, 4, 3), v = BF(w1, 9, 8); + + if (dim == 3) { sbs(s, "invalid_smp_lod_mode"); return; } + if (fmt == 3) { sbs(s, "invalid_smp_data_type"); return; } + sbs(s, pred3[BF(w1, 26, 24)]); + sbf(s, "smp%ud%s", dim + 1, var[v]); + if (BIT(w1, 23)) sbs(s, ".skipinv"); + if (BIT(w1, 20)) sbs(s, ".syncs"); + emit_repeat(s, w1); + sbs(s, " "); + dst_noalias = 1; /* SMP names its destination temp */ + emit_dst(s, BIT(w1, 7) ? 2 : 0, BF(w0, 27, 21)); + dst_noalias = 0; + emit_dstmask(s, w1); + sbs(s, ", "); + emit_src(s, s0bank[(BIT(w1, 18) << 1) | BIT(w1, 2)], BF(w0, 20, 14), + 0, 0, s0fmt[fmt]); + sbs(s, ", "); + emit_src(s, src_bank(w0, w1, 1), BF(w0, 13, 7), 0, 0, NULL); + if (v) { + sbs(s, ", "); + emit_src(s, src_bank(w0, w1, 2), BF(w0, 6, 0), 0, 0, NULL); + } + sbf(s, ", drc%u", BF(w1, 1, 0)); +} + +/* group 0x18: unsigned 8- and 16-bit dot products. Register fields are six + * bits wide here, the seventh bit being a .c10 flag -- except in the constant + * and immediate banks, which use all seven -- so the indexed form takes the + * compact layout. w1[21:20] == 3 selects the u16 variant and w1[24] the + * four-component form; the alpha co-issue is only present for a dot3 with + * w1[10:9] != 3, and the alpha scale is printed separately when it is not. */ +/* Shared operand frame: destination w0[27:21], sources w0[20:14], w0[13:7] + * and w0[6:0], banks in w0[31:28] plus the extension bits w1[19], w1[17] and + * w1[16], and the modifier block w1[23:12]. IMA8, FPMA and the dot products + * spend the top bit of every register field on a .c10 flag instead -- except + * in the constant and immediate banks, which keep all seven -- which also + * makes their indexed form take the compact layout. */ +static void emit_frame_src(struct sb *s, int bank, unsigned num, int c10, + int neg, const char *suf) +{ + if (c10 && bank != 5 && bank != 6) { + src_tail = BIT(num, 6) ? ".c10" : NULL; + num = BF(num, 5, 0); + } + emit_src_f(s, bank, num, neg, 0, suf, c10); + src_tail = NULL; +} + +/* source 0 of the frame; FIRH and FIRV share source 1's bank for it */ +static void emit_frame_src0(struct sb *s, uint32_t w0, uint32_t w1, int bank, + int c10, int neg, const char *suf) +{ + emit_frame_src(s, bank, BF(w0, 20, 14), c10, neg, suf); +} + +static void emit_frame_dst(struct sb *s, uint32_t w0, uint32_t w1, int c10) +{ + int bank = (BIT(w1, 19) << 2) | BF(w1, 1, 0), lo = 0; + unsigned n = BF(w0, 27, 21), alias = c10 ? 60 : 124; + + if (c10 && bank != 5 && bank != 6) { lo = BIT(n, 6); n = BF(n, 5, 0); } + if (!bank && n >= alias) sbf(s, "i%u", n - alias); + else if (bank == 3) emit_indexed(s, n, c10); + else emit_dst(s, bank, n); + if (lo) sbs(s, ".c10"); +} + +/* skipinv / nosched / end / repeat, in the order this group prints them */ +static void emit_frame_mods(struct sb *s, uint32_t w1, unsigned rep) +{ + if (BIT(w1, 23)) sbs(s, ".skipinv"); + if (BIT(w1, 22)) sbs(s, ".nosched"); + if (BIT(w1, 18)) sbs(s, ".end"); + if (rep) sbf(s, ".repeat%u", rep + 1); +} + +/* group 0x18: unsigned 8- and 16-bit dot products. w1[21:20] == 3 selects the + * u16 variant and w1[24] the four-component form; the alpha co-issue is only + * present for a dot3 with w1[10:9] != 3, and the alpha scale is then printed + * as part of it rather than as an operand of its own. */ +static void dis_dot(struct sb *s, uint32_t w0, uint32_t w1) +{ + static const char *const scale[4] = { "x1", "x2", "x4", "x8" }; + static const char *const sel[4] = { "zero", "s1a", "s2a", "" }; + static const char *const aop[4] = { "aadd", "asub", "aintrp1", "aintrp2" }; + unsigned o1 = BF(w1, 21, 20), o3 = BF(w1, 10, 9), a = BF(w1, 5, 4); + int u16 = o1 == 3, dot4 = BIT(w1, 24); + int co = !dot4 && !u16 && o3 != 3; + + sbs(s, pred2[BF(w1, 26, 25)]); + sbf(s, "u%ddot%d%s", u16 ? 16 : 8, dot4 ? 4 : 3, BIT(w1, 6) ? "off" : ""); + emit_frame_mods(s, w1, BF(w1, 14, 12)); + sbs(s, " "); + emit_frame_dst(s, w0, w1, 1); + sbf(s, ", C%s", scale[BF(w1, 8, 7)]); + if (dot4 || o3 == 3) sbf(s, ", A%s", scale[BF(w0, 17, 16)]); + sbs(s, ", "); + emit_frame_src(s, src_bank(w0, w1, 1), BF(w0, 13, 7), 1, 0, + BIT(w1, 11) ? ".comp" : NULL); + sbs(s, ", "); + emit_frame_src(s, src_bank(w0, w1, 2), BF(w0, 6, 0), 1, 0, + BIT(w1, 15) ? ".comp" : NULL); + if (!co) return; + sbf(s, "+%s ", aop[a]); + if (a >= 2) { + emit_frame_src0(s, w0, w1, BIT(w1, 2) ? 2 : 0, 1, 0, NULL); + sbf(s, ", %s, %s%s", sel[o1], sel[o3], BIT(w1, 3) ? ".comp" : ""); + } else { + sbf(s, "A%s, %s%s, s1a%s, %s%s, s2a", scale[BF(w0, 17, 16)], + sel[o1], BIT(w1, 2) ? ".comp" : "", + BIT(w0, 15) ? ".comp" : "", + sel[o3], BIT(w1, 3) ? ".comp" : ""); + } +} + +/* groups 0x13 and 0x19: IMA8 and FPMA share one encoding -- three sources in + * the c10 frame followed by six operand selectors, each with its own .comp. */ +static void dis_ima8(struct sb *s, uint32_t w0, uint32_t w1, const char *name) +{ + static const char *const sel5[4] = { "s0", "s1", "s0a", "s1a" }; + + sbs(s, pred2[BF(w1, 26, 25)]); + sbs(s, name); + emit_frame_mods(s, w1, BF(w1, 14, 12)); + if (BIT(w1, 11)) sbs(s, ".sat"); + sbs(s, " "); + emit_frame_dst(s, w0, w1, 1); + sbs(s, ", "); + emit_frame_src0(s, w0, w1, BIT(w1, 2) ? 2 : 0, 1, BIT(w1, 3), NULL); + sbs(s, ", "); + emit_frame_src(s, src_bank(w0, w1, 1), BF(w0, 13, 7), 1, 0, NULL); + sbs(s, ", "); + emit_frame_src(s, src_bank(w0, w1, 2), BF(w0, 6, 0), 1, 0, NULL); + sbf(s, ", %s%s", sel5[BF(w1, 21, 20)], BIT(w1, 10) ? ".comp" : ""); + sbf(s, ", %s%s", BIT(w1, 5) ? "s1a" : "s1", BIT(w1, 24) ? ".comp" : ""); + sbf(s, ", %s%s", BIT(w1, 4) ? "s2a" : "s2", BIT(w1, 15) ? ".comp" : ""); + sbf(s, ", %s%s", BIT(w1, 9) ? "s1a" : "s0a", BIT(w1, 6) ? ".comp" : ""); + sbf(s, ", s1a%s", BIT(w1, 7) ? ".comp" : ""); + sbf(s, ", s2a%s", BIT(w1, 8) ? ".comp" : ""); +} + +/* group 0x16: ADIF, BILIN and FIRV, picked by w1[21:20]; and group 0x17, + * FIRH. All four use the plain seven-bit frame. FIRH and FIRV have no bank + * of their own for source 0 and borrow source 1's. */ +static void dis_adif(struct sb *s, uint32_t w0, uint32_t w1) +{ + static const char *const fmt[4] = { "u8", "s8", "o8", NULL }; + unsigned sel = BF(w1, 21, 20), f = BF(w1, 9, 8); + int s1bank = src_bank(w0, w1, 1); + + if (sel == 1) { sbs(s, "invalid_adiffirvbilin_opsel"); return; } + if (sel >= 2 && f == 3) { + sbs(s, sel == 2 ? "invalid_bilin_source_format" + : "invalid_firv_flag"); + return; + } + sbs(s, pred2[BF(w1, 26, 25)]); + sbs(s, sel == 2 ? "bilin" : sel == 3 ? "firv" : + BIT(w1, 6) ? "adifsum" : "adif"); + emit_frame_mods(s, w1, BF(w1, 14, 12)); + sbs(s, " "); + if (sel == 3 && !BIT(w1, 15)) sbs(s, "!"); + emit_frame_dst(s, w0, w1, 0); + if (sel || BIT(w1, 6)) { + sbs(s, ", "); + emit_frame_src0(s, w0, w1, + sel == 3 ? s1bank : BIT(w1, 2) ? 2 : 0, 0, 0, + sel ? NULL : BIT(w1, 15) ? ".high" : ".low"); + } + sbs(s, ", "); + emit_frame_src(s, s1bank, BF(w0, 13, 7), 0, 0, NULL); + sbs(s, ", "); + emit_frame_src(s, src_bank(w0, w1, 2), BF(w0, 6, 0), 0, 0, NULL); + if (sel < 2) return; + sbf(s, ", %s", fmt[f]); + if (sel == 2) { + sbf(s, ", %s", BIT(w1, 24) ? "interleaved" : "planar"); + sbf(s, ", %s", BIT(w1, 11) ? "src23" : "src01"); + sbf(s, ", %s", BIT(w1, 10) ? "dst23" : "dst01"); + if (BIT(w1, 15)) sbs(s, ", rnd"); + } else { + if (BIT(w1, 24)) sbs(s, ", +src2.2"); + if (BIT(w1, 7)) sbs(s, ", iadd"); + if (BIT(w1, 10)) sbs(s, ", scale"); + } +} + +static void dis_firh(struct sb *s, uint32_t w0, uint32_t w1) +{ + static const char *const fmt[4] = { "u8", "s8", "o8", NULL }; + static const char *const mode[4] = { "rep", "msc", "mdc", "reserved" }; + unsigned f = BF(w1, 9, 8); + int s1bank = src_bank(w0, w1, 1); + int imm = BIT(w1, 10) | (BIT(w1, 11) << 1) | (BIT(w1, 14) << 2) | + (BIT(w1, 15) << 3) | (BIT(w1, 24) << 4); + + if (f == 3) { sbs(s, "invalid_firh_source_format"); return; } + sbs(s, pred2[BF(w1, 26, 25)]); + sbs(s, "firh"); + emit_frame_mods(s, w1, BF(w1, 13, 12)); + sbs(s, " "); + emit_frame_dst(s, w0, w1, 0); + sbs(s, ", "); + emit_frame_src0(s, w0, w1, s1bank, 0, 0, NULL); + sbs(s, ", "); + emit_frame_src(s, s1bank, BF(w0, 13, 7), 0, 0, NULL); + sbs(s, ", "); + emit_frame_src(s, src_bank(w0, w1, 2), BF(w0, 6, 0), 0, 0, NULL); + sbf(s, ", %s, %s, fcs%u, #%d, #%u", fmt[f], mode[BF(w1, 7, 6)], + BF(w1, 5, 3), imm >= 16 ? imm - 32 : imm, BF(w1, 21, 20)); +} + +static void dis_bitwise(struct sb *s, uint32_t w0, uint32_t w1, const char *name) +{ + int b2 = src_bank(w0, w1, 2); + sbs(s, pred3[BF(w1, 26, 24)]); + sbs(s, name); + if (BIT(w1, 23)) sbs(s, ".skipinv"); + if (BIT(w1, 20)) sbs(s, ".syncs"); + if (BIT(w1, 18)) sbs(s, ".end"); + if (BIT(w1, 2)) sbs(s, ".partial"); + emit_repeat(s, w1); + sbs(s, " "); + emit_dst(s, (BIT(w1, 19) << 2) | BF(w1, 1, 0), BF(w0, 27, 21)); + emit_dstmask(s, w1); + sbs(s, ", "); + emit_src(s, src_bank(w0, w1, 1), BF(w0, 13, 7), 0, 0, NULL); + sbs(s, ", "); + if (BIT(w1, 11)) sbs(s, "~"); + if (b2 == 6) { + uint32_t imm = BF(w0, 6, 0) | (BF(w0, 20, 14) << 7) | + (BF(w1, 5, 4) << 14); + unsigned rot = BF(w1, 10, 6); + sbf(s, "#0x%08X", rot ? (imm << rot) | (imm >> (32 - rot)) : imm); + } else { + emit_src(s, b2, BF(w0, 6, 0), 0, 0, NULL); + } +} + +/* group 0x0E: RLP */ +static void dis_rlp(struct sb *s, uint32_t w0, uint32_t w1) +{ + sbs(s, pred3[BF(w1, 26, 24)]); + sbs(s, "rlp"); + if (BIT(w1, 23)) sbs(s, ".skipinv"); + if (BIT(w1, 20)) sbs(s, ".syncs"); + if (BIT(w1, 18)) sbs(s, ".end"); + if (BIT(w1, 2)) sbs(s, ".partial"); + emit_repeat(s, w1); + sbs(s, " "); + emit_dst(s, (BIT(w1, 19) << 2) | BF(w1, 1, 0), BF(w0, 27, 21)); + emit_dstmask(s, w1); + sbs(s, ", "); + emit_src(s, src_bank(w0, w1, 1), BF(w0, 13, 7), 0, 0, NULL); + sbf(s, ", p%u", BF(w0, 20, 14)); +} + +/* groups 0x14/0x15: IMA16 / IMAE */ +static void dis_imae(struct sb *s, uint32_t w0, uint32_t w1, int is_imae) +{ + static const char *const sz[4] = { "z16", "u16", "u32", NULL }; + static const char *const sgn[4] = { ".u", ".s", ".usat", ".ssat" }; + unsigned f = BF(w1, 7, 6); + unsigned rp = BF(w1, 14, 12) + 1; + unsigned rs = is_imae ? BF(w1, 5, 3) : BF(w1, 4, 3); + int b2 = src_bank(w0, w1, 2); + if (is_imae && !sz[f]) { sbs(s, "invalid_imae_src2_type"); return; } + sbs(s, pred2[BF(w1, 26, 25)]); + sbs(s, is_imae ? "imae" : "ima16"); + if (BIT(w1, 23)) sbs(s, ".skipinv"); + if (BIT(w1, 22)) sbs(s, ".nosched"); + if (BIT(w1, 18)) sbs(s, ".end"); + if (rp > 1) sbf(s, ".repeat%u", rp); + sbs(s, sgn[(BIT(w1, 10) << 1) | BIT(w1, 11)]); + if (rs) sbf(s, ".rs%u", rs); + sbs(s, " "); + emit_dst(s, (BIT(w1, 19) << 2) | BF(w1, 1, 0), BF(w0, 27, 21)); + if (!is_imae && BIT(w1, 24)) sbs(s, ".abs"); + sbs(s, ", "); + if (is_imae) { + emit_src0(s, w0, w1, 0, 0, BIT(w1, 24) ? ".high" : ".low"); + sbs(s, ", "); + emit_src(s, src_bank(w0, w1, 1), BF(w0, 13, 7), 0, 0, + BIT(w1, 21) ? ".high" : ".low"); + sbs(s, ", "); + emit_src(s, b2, BF(w0, 6, 0), 0, 0, + f == 2 || f == 3 ? NULL : + (BIT(w1, 20) ? ".high" : ".low")); + sbf(s, ", %s", sz[f]); + if (BIT(w1, 8)) sbs(s, ", coe"); + if (BIT(w1, 9)) sbs(s, ", cie"); + if (BIT(w1, 8) || BIT(w1, 9)) sbf(s, ", i%u", BIT(w1, 15)); + } else { + static const char *const i16[4] = { "u16", "u8", "s8", "o8" }; + emit_src0(s, w0, w1, 0, 0, NULL); + sbs(s, ", "); + emit_src(s, src_bank(w0, w1, 1), BF(w0, 13, 7), 0, 0, + BF(w1, 7, 6) ? (BIT(w1, 20) ? ".high" : ".low") : NULL); + sbs(s, ", "); + emit_src(s, b2, BF(w0, 6, 0), BIT(w1, 21), 0, NULL); + if (BF(w1, 9, 8)) sbs(s, BIT(w1, 5) ? ".high" : ".low"); + sbf(s, ", %s, %s", i16[BF(w1, 7, 6)], i16[BF(w1, 9, 8)]); + } +} + +/* groups 0x1D/0x1E: LD and ST */ +static void dis_ldst(struct sb *s, uint32_t w0, uint32_t w1, int is_store) +{ + static const char *const mode[4] = { "a", "l", "t", "?" }; + static const char *const size[4] = { "d", "w", "b", "?" }; + unsigned md = BF(w1, 11, 10), sz = BF(w1, 5, 4); + int base_bank = (BIT(w1, 18) << 1) | BIT(w1, 2); + int ofs_bank = src_bank(w0, w1, 1); + + if (md == 3 || sz == 3) { sbs(s, "invalid_ldst_data_size"); return; } + sbs(s, pred3[BF(w1, 26, 24)]); + sbf(s, "%s%s%s", is_store ? "st" : "ld", mode[md], size[sz]); + if (BIT(w1, 23)) sbs(s, ".skipinv"); + if (BIT(w1, 20)) sbs(s, ".syncs"); + if (BIT(w1, 19)) sbs(s, ".bpcache"); + if (BIT(w1, 1)) sbs(s, ".fcfill"); + if (is_store && BIT(w1, 6)) sbs(s, ".rangeenable"); + if (BIT(w1, 21)) { + unsigned n = BF(w1, 15, 12) + 1; + if (n > 1) sbf(s, ".repeat%u", n); + } else { + sbf(s, ".fetch%u", BF(w1, 15, 12) + 1); + } + sbs(s, " "); + if (!is_store) { + sbf(s, "%s%u, ", BIT(w1, 7) ? "pa" : "r", BF(w0, 27, 21)); + } + sbs(s, "["); + emit_ext(s, base_bank, BF(w0, 20, 14)); + sbs(s, ","); + if (BF(w1, 9, 8) != 1 && BF(w1, 9, 8) != 2) + sbs(s, BIT(w1, 3) ? "-" : "+"); + else if (BF(w1, 9, 8) == 1) + sbs(s, BIT(w1, 3) ? "--" : "++"); + if (ofs_bank == 6) + sbf(s, "#%u", BF(w0, 13, 7)); + else + emit_src(s, ofs_bank, BF(w0, 13, 7), 0, 0, NULL); + if (BF(w1, 9, 8) == 2) sbs(s, BIT(w1, 3) ? "--" : "++"); + sbs(s, "]"); + if (is_store) { + int b = src_bank(w0, w1, 2); + sbs(s, ", "); + emit_src(s, b, b == 6 ? BF(w0, 6, 0) : BF(w0, 6, 0), 0, 0, NULL); + } else { + if (BIT(w1, 6)) { + int b = src_bank(w0, w1, 2); + sbs(s, ", "); + emit_src(s, b, BF(w0, 6, 0), 0, 0, NULL); + } + sbf(s, ", drc%u", BF(w1, 0, 0)); + } +} + +/* group 0x1F, sub-group 21:20 == 0: flow control */ +static void dis_flow(struct sb *s, uint32_t w0, uint32_t w1, enum usp_opcode op) +{ + if (BIT(w1, 22)) { sbs(s, "invalid_other_op2"); return; } + if (op == USP_NOP) { + sbs(s, "nop"); + if (BIT(w1, 23)) sbs(s, ".syncs"); + if (BIT(w1, 11)) sbs(s, ".nosched"); + if (BIT(w1, 18)) sbs(s, ".end"); + if (BIT(w0, 0)) sbs(s, ".tog"); + return; + } + if (op < USP_BA || op > USP_NOP) { + sbs(s, "invalid_flow_control_op2"); + return; + } + sbs(s, pred3[BF(w1, 26, 24)]); + switch (op) { + case USP_BA: + case USP_BR: + sbs(s, op == USP_BA ? "ba" : "br"); + if (BIT(w1, 23)) sbs(s, ".synce"); + if (BIT(w1, 9)) sbs(s, ".savelink"); + if (op == USP_BR && BIT(w0, 11)) /* relative, signed */ + sbf(s, " -#0x%08X", 0x1000 - BF(w0, 11, 0)); + else + sbf(s, " #0x%08X", BF(w0, 11, 0)); + break; + case USP_LAPC: + sbs(s, "lapc"); + if (BIT(w1, 23)) sbs(s, ".synce"); + break; + case USP_SETL: + sbs(s, "mov pclink, "); + if (src_bank(w0, w1, 1) == 6) + sbf(s, "#0x%08X", BF(w0, 13, 7)); + else + emit_src(s, src_bank(w0, w1, 1), BF(w0, 13, 7), 0, 0, NULL); + break; + case USP_SAVL: + sbs(s, "mov "); + emit_dst(s, (BIT(w1, 19) << 2) | BF(w1, 1, 0), BF(w0, 27, 21)); + sbs(s, ", pclink"); + break; + case USP_NOP: + break; + default: + sbs(s, "invalid_flow_control_op2"); + } +} + +/* group 0x1F, sub-group 21:20 == 1: MOE control */ +static void dis_moe(struct sb *s, uint32_t w0, uint32_t w1, enum usp_opcode op) +{ + unsigned i; + switch (op) { + case USP_SMOA: { + static const char *const wm[4] = { "none", "repeat", "clamp", + "mirror" }; + sbs(s, "smoa "); + sbf(s, "#%d, ", s10(BF(w0, 31, 30) | (BF(w1, 7, 0) << 2))); + sbf(s, "#%d, ", s10(BF(w0, 29, 20))); + sbf(s, "#%d, ", s10(BF(w0, 19, 10))); + sbf(s, "#%d, ", s10(BF(w0, 9, 0))); + for (i = 0; i < 4; i++) + sbf(s, "%s%s", wm[BF(w1, 15 - i * 2, 14 - i * 2)], + i < 3 ? ", " : ""); + break; + } + case USP_SMR: + sbf(s, "smr #%u, #%u, #%u, #%u", BF(w1, 15, 4), + BF(w0, 31, 24) | (BF(w1, 3, 0) << 8), + BF(w0, 23, 12), BF(w0, 11, 0)); + break; + case USP_SMLSI: + sbs(s, "smlsi "); + for (i = 0; i < 4; i++) { + unsigned b = BF(w0, 31 - i * 8, 24 - i * 8); + if (BIT(w1, 3 - i)) + sbf(s, "swizzle(%c%c%c%c), ", swz[BF(b, 1, 0)], + swz[BF(b, 3, 2)], swz[BF(b, 5, 4)], + swz[BF(b, 7, 6)]); + else sbf(s, "#%d, ", (int)(signed char)b); + } + for (i = 0; i < 4; i++) + sbf(s, "%s, ", BIT(w1, 3 - i) ? "swizzlemode" + : "incrementmode"); + sbf(s, "#%u, #%u, #%u", BF(w1, 15, 12) * 4, + BF(w1, 11, 8) * 4, BF(w1, 7, 4) * 4); + break; + case USP_SMBO: + sbf(s, "smbo #%u, #%u, #%u, #%u", + BF(w1, 15, 4), BF(w0, 31, 24) | (BF(w1, 3, 0) << 8), + BF(w0, 23, 12), BF(w0, 11, 0)); + break; + case USP_IMO: + sbf(s, "imo #%d, #%d, #%d, #%d", moe_s7(BF(w1, 15, 4)), + moe_s7(BF(w0, 31, 24) | (BF(w1, 3, 0) << 8)), + moe_s7(BF(w0, 23, 12)), moe_s7(BF(w0, 11, 0))); + break; + case USP_SETFC: + sbf(s, "setfc #%u, #%u", BIT(w0, 0), BIT(w0, 8)); + break; + default: + sbs(s, "invalid_moe_op2"); + } +} + +/* group 0x1F, sub-group 21:20 == 2: memory / misc */ +static void dis_misc(struct sb *s, uint32_t w0, uint32_t w1, enum usp_opcode op) +{ + static const char *const idfp[4] = { "bif", "pixelbe", "reserved(=2)", "reserved(=3)" }; + uint32_t imm; + switch (op) { + case USP_IDF: + sbf(s, "idf drc%u, %s", BF(w1, 1, 0), idfp[BF(w1, 15, 14)]); + break; + case USP_WDF: + sbf(s, "wdf drc%u", BF(w1, 1, 0)); + break; + case USP_EMIT: { + unsigned tgt = BF(w1, 15, 14); + unsigned sb = BF(w0, 27, 22) | (BF(w1, 8, 3) << 6) | + (BF(w1, 23, 22) << 12); + int s0 = 1, s1 = 1, s2 = 1, wimm = 1; + if (tgt == 3) { sbs(s, "invalid_emit_target"); break; } + if (tgt == 0) { + sbs(s, BIT(w0, 26) ? "emitpix1" : "emitpix2"); + if (BIT(w0, 26)) { s2 = 0; wimm = 0; } + } else if (tgt == 1) { + s0 = 0; wimm = 0; + switch (BF(w1, 13, 12)) { + case 0: sbs(s, "emitst"); s2 = 0; break; + case 1: sbs(s, "emitvtx"); s1 = 0; break; + case 2: sbs(s, "emitprimitive"); wimm = 1; break; + default: sbs(s, "invalid_emit_mte_control"); break; + } + if (BF(w1, 13, 12) == 3) break; + } else { + sbs(s, "emitpds"); + } + if (BIT(w1, 18)) sbs(s, ".end"); + if (tgt == 2) { + if (BIT(w1, 13)) sbs(s, ".tasks"); + if (BIT(w1, 12)) sbs(s, ".taske"); + } + if (BIT(w0, 21)) sbs(s, ".freep"); + sbf(s, " #%u /* incp */", BF(w1, 1, 0)); + if (s0) { + sbs(s, ", "); + emit_src0_ext(s, w0, w1); + } + if (s1) { + sbs(s, ", "); + emit_src(s, src_bank(w0, w1, 1), BF(w0, 13, 7), 0, 0, NULL); + } + if (s2) { + sbs(s, ", "); + emit_src(s, src_bank(w0, w1, 2), BF(w0, 6, 0), 0, 0, NULL); + } + if (wimm) sbf(s, ", #0x%08X", sb); + break; + } + case USP_LIMM: + imm = BF(w0, 20, 0) | (BF(w1, 8, 4) << 21) | (BF(w1, 17, 12) << 26); + sbs(s, pred3[BF(w1, 11, 9)]); + sbs(s, "mov"); + if (BIT(w1, 23)) sbs(s, ".skipinv"); + if (BIT(w1, 22)) sbs(s, ".nosched"); + if (BIT(w1, 18)) sbs(s, ".end"); + sbs(s, " "); + emit_dst(s, (BIT(w1, 19) << 2) | BF(w1, 1, 0), BF(w0, 27, 21)); + sbf(s, ", #0x%08X", imm); + break; + case USP_LOCK: + case USP_RELEASE: + if (BF(w0, 7, 4)) sbs(s, "invalid_mutex_number"); + else sbf(s, "%s #0", op == USP_LOCK ? "lock" : "release"); + break; + case USP_LDR: + case USP_STR: { + int b2 = src_bank(w0, w1, 2); + unsigned n = b2 == 6 ? src2_imm(w0) : BF(w0, 6, 0); + sbs(s, op == USP_LDR ? "ldr" : "str"); + if (BIT(w1, 23)) sbs(s, ".skipinv"); + sbs(s, " "); + if (op == USP_LDR) { + sbf(s, "%s%u, ", BIT(w1, 7) ? "pa" : "r", BF(w0, 27, 21)); + emit_src(s, b2, n, 0, 0, NULL); + sbf(s, ", drc%u", BF(w1, 1, 0)); + } else { + emit_src(s, b2, n, 0, 0, NULL); + sbs(s, ", "); + emit_src(s, src_bank(w0, w1, 1), BF(w0, 13, 7), 0, 0, NULL); + } + break; + } + case USP_WOP: + sbf(s, "wop #%u", BF(w0, 1, 0)); + break; + default: + sbs(s, "invalid_opcode"); + } +} + +/* group 0x1F, sub-group 21:20 == 3: visibility test / coefficient ops */ +static void dis_vistest(struct sb *s, uint32_t w0, uint32_t w1) +{ + static const char *const fb[4] = { "fb", "nfb", "optdwd", "optdwd" }; + unsigned sub = BF(w1, 26, 24); + if (sub == 0) { + if (BIT(w1, 15)) { + sbs(s, "ptoff"); + if (BIT(w1, 18)) sbs(s, ".end"); + return; + } + sbs(s, "pcoeff"); + if (BIT(w1, 18)) sbs(s, ".end"); + sbs(s, " "); + emit_src0_ext(s, w0, w1); + sbs(s, ", "); + emit_src(s, src_bank(w0, w1, 1), BF(w0, 13, 7), 0, 0, NULL); + sbs(s, ", "); + emit_src(s, src_bank(w0, w1, 2), BF(w0, 6, 0), 0, 0, NULL); + return; + } + if (sub != 1 && sub != 3) { sbs(s, "invalid_vistest_op2"); return; } + sbs(s, pred2[BF(w1, 10, 9)]); + sbs(s, sub == 1 ? "atst8" : "depthf"); + if (BIT(w1, 23)) sbs(s, ".syncs"); + if (BIT(w1, 11)) sbs(s, ".nosched"); + if (BIT(w1, 18)) sbs(s, ".end"); + if (sub == 1 && BIT(w1, 22)) sbs(s, ".c10"); + sbs(s, " "); + if (sub == 1) + emit_dst(s, BF(w1, 1, 0), BF(w0, 27, 21)); + else + emit_src0_ext(s, w0, w1); + sbs(s, ", "); + if (sub == 1) { + sbf(s, "%sp%u, ", BIT(w1, 6) ? "" : "!", BF(w1, 8, 7)); + emit_src0_ext(s, w0, w1); + sbs(s, ", "); + } + emit_src(s, src_bank(w0, w1, 1), BF(w0, 13, 7), 0, 0, NULL); + sbs(s, ", "); + emit_src(s, src_bank(w0, w1, 2), BF(w0, 6, 0), 0, 0, NULL); + if (BIT(w1, 5)) sbs(s, ", twosided"); + sbf(s, ", %s", fb[BF(w1, 4, 3)]); +} + +/* ------------------------------------------------------------------ */ + +int usse_disasm(const struct usse_insn *in, char *buf, size_t buflen) +{ + uint32_t w0 = in->w0, w1 = in->w1; + unsigned g = BF(w1, 31, 27); + enum usp_opcode op = usse_classify(in); + struct sb s; + + sb_init(&s, buf, buflen); + + switch (g) { + case 0x00: dis_mad(&s, w0, w1); break; + case 0x01: dis_rcp(&s, w0, w1); break; + case 0x02: dis_dp(&s, w0, w1); break; + case 0x03: + switch (BF(w1, 10, 9)) { + case 0: dis_2src(&s, w0, w1, "fmin", 2); break; + case 1: dis_2src(&s, w0, w1, "fmax", 2); break; + default: sbs(&s, "invalid_float_op2"); + } + break; + case 0x04: + switch (BF(w1, 10, 9)) { + case 0: dis_2src(&s, w0, w1, "fdsx", 2); break; + case 1: dis_2src(&s, w0, w1, "fdsy", 2); break; + default: sbs(&s, "invalid_float_op2"); + } + break; + case 0x05: dis_movc(&s, w0, w1); break; + case 0x07: dis_efo(&s, w0, w1); break; + case 0x06: + if (BF(w1, 10, 9)) sbs(&s, "invalid_float_op2"); + else dis_fmad16(&s, w0, w1); + break; + case 0x08: dis_pck(&s, w0, w1); break; + case 0x10: dis_sop2(&s, w0, w1); break; + case 0x11: + if (!BIT(w1, 21)) dis_sop3(&s, w0, w1); + else if (BIT(w1, 10)) dis_lrp2(&s, w0, w1); + else dis_lrp1(&s, w0, w1); + break; + case 0x12: dis_sopwm(&s, w0, w1); break; + case 0x09: dis_test(&s, w0, w1); break; + case 0x0A: dis_bitwise(&s, w0, w1, BIT(w1, 3) ? "or" : "and"); break; + case 0x0B: + if (BIT(w1, 3)) sbs(&s, "invalid_bitwise_op2"); + else dis_bitwise(&s, w0, w1, "xor"); + break; + case 0x0C: dis_bitwise(&s, w0, w1, BIT(w1, 3) ? "rol" : "shl"); break; + case 0x0D: dis_bitwise(&s, w0, w1, BIT(w1, 3) ? "asr" : "shr"); break; + case 0x0E: dis_rlp(&s, w0, w1); break; + case 0x0F: dis_testmask(&s, w0, w1); break; + case 0x14: dis_imae(&s, w0, w1, 0); break; + case 0x15: dis_imae(&s, w0, w1, 1); break; + case 0x13: dis_ima8(&s, w0, w1, "ima8"); break; + case 0x16: dis_adif(&s, w0, w1); break; + case 0x17: dis_firh(&s, w0, w1); break; + case 0x18: dis_dot(&s, w0, w1); break; + case 0x19: dis_ima8(&s, w0, w1, "fpma"); break; + case 0x1C: dis_smp(&s, w0, w1); break; + case 0x1D: dis_ldst(&s, w0, w1, 0); break; + case 0x1E: dis_ldst(&s, w0, w1, 1); break; + case 0x1F: + switch (BF(w1, 21, 20)) { + case 0: dis_flow(&s, w0, w1, op); break; + case 1: dis_moe(&s, w0, w1, op); break; + case 2: dis_misc(&s, w0, w1, op); break; + default: dis_vistest(&s, w0, w1); break; + } + break; + default: + /* Opcode groups whose operand encoding is not modelled are + * emitted verbatim so that the assembler can reproduce them. */ + sbf(&s, ".word 0x%08X, 0x%08X /* %s */", w1, w0, + usp_opcode_name[op]); + return 1; + } + return s.trunc ? -1 : 0; +} diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/usse.h mesa-26.2.2/src/gallium/drivers/sgx/usse.h --- mesa-26.2.2.orig/src/gallium/drivers/sgx/usse.h 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/usse.h 2026-09-08 10:57:36.684127499 +0200 @@ -0,0 +1,63 @@ +/* + * usse.h - shared definitions for the USSE (PowerVR SGX Series5) toolchain + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#ifndef USSE_H +#define USSE_H + +#include +#include + +/* A USSE instruction is 64 bit, stored little-endian as word0 then word1. + * The archived encoding tables address bits as "1:31-27" = w1 bits 31..27. */ +struct usse_insn { + uint32_t w0; + uint32_t w1; +}; + +#define BF(w, hi, lo) (((w) >> (lo)) & ((1u << ((hi) - (lo) + 1)) - 1)) +#define BIT(w, b) (((w) >> (b)) & 1u) + +/* USP_OPCODE enumeration, upstream/usse-docs/USP_OPCODE.txt */ +enum usp_opcode { + USP_INVALID = 0, USP_MAD, USP_ADM, USP_MSA, USP_FRC, USP_RCP, USP_RSQ, + USP_LOG, USP_EXP, USP_DP, USP_DDP, USP_DDPC, USP_MIN, USP_MAX, USP_DSX, + USP_DSY, USP_MOVC, USP_FMAD16, USP_EFO, USP_PCKUNPCK, USP_TEST, USP_AND, + USP_OR, USP_XOR, USP_SHL, USP_ROL, USP_SHR, USP_ASR, USP_RLP, + USP_TESTMASK, USP_SOP2, USP_SOP3, USP_SOPWM, USP_IMA8, USP_IMA16, + USP_IMAE, USP_ADIF, USP_BILIN, USP_FIRV, USP_FIRH, USP_DOT3, USP_DOT4, + USP_FPMA, USP_SMP, USP_SMPBIAS, USP_SMPREPLACE, USP_SMPGRAD, USP_LD, + USP_ST, USP_BA, USP_BR, USP_LAPC, USP_SETL, USP_SAVL, USP_NOP, USP_SMOA, + USP_SMR, USP_SMLSI, USP_SMBO, USP_IMO, USP_SETFC, USP_IDF, USP_WDF, + USP_SETM, USP_EMIT, USP_LIMM, USP_LOCK, USP_RELEASE, USP_LDR, USP_STR, + USP_WOP, USP_PCOEFF, USP_PTOFF, USP_ATST8, USP_DEPTHF, + USP_OPCODE_COUNT /* 75 */ +}; + +extern const char *const usp_opcode_name[USP_OPCODE_COUNT]; + +/* Classify a raw instruction word pair into a USP_OPCODE. Returns + * USP_INVALID for encodings the tables mark invalid. */ +enum usp_opcode usse_classify(const struct usse_insn *in); + +/* Disassemble into buf (vendor-compatible text). Returns 0 on success. */ +int usse_disasm(const struct usse_insn *in, char *buf, size_t buflen); + +/* Assemble one line of text. Returns 0 on success, -1 on parse error with + * an English message in err. */ +int usse_asm(const char *line, struct usse_insn *out, char *err, size_t errlen); + +/* Register banks as the vendor disassembler names them. Source operands and + * destination operands use different orderings of the same 3-bit field. */ +enum usse_bank { + BK_TEMP = 0, BK_OUTPUT, BK_PRIMATTR, BK_SECATTR, + BK_INDEX, BK_SPECIAL, BK_IMMEDIATE, BK_FPINTERNAL, + BK_IREG /* index-register write, dest only */ +}; + +extern const char *const usse_bank_name[9]; + +#endif /* USSE_H */ diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/usse_backend.h mesa-26.2.2/src/gallium/drivers/sgx/usse_backend.h --- mesa-26.2.2.orig/src/gallium/drivers/sgx/usse_backend.h 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/usse_backend.h 2026-09-08 10:57:36.683723859 +0200 @@ -0,0 +1,71 @@ +/* + * usse_backend.h - backend-private entry point and register map + * + * struct uir_codegen_result in usse_ir.h now carries the register counts and + * the literal pool, which is everything a driver needs to run the code. This + * header adds what only tools want: the full binding-to-register map, the + * intended instruction listing, and the unlowered-instruction report. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#ifndef USSE_BACKEND_H +#define USSE_BACKEND_H + +#include "usse_ir.h" +#include "usse_isa.h" + +#define USSE_MAX_POOL UIR_MAX_POOL + +/* Retained for callers written against the earlier backend-only pool; the + * same entries come out of uir_codegen_result.pool with the raw dwords. */ +struct usse_pool_entry { + float v[4]; + uint32_t sa_base; +}; + +struct usse_map_entry { + enum uir_regclass cls; + uint32_t index; + uint32_t base; /* first hardware register */ + uint32_t count; /* always 4 in this backend */ +}; + +struct usse_layout { + struct usse_map_entry *map; + size_t nmap, map_cap; + + struct usse_pool_entry pool[USSE_MAX_POOL]; + size_t npool; + uint32_t pool_base; /* first sa register of the constant pool */ + + /* Instructions the backend could not lower. Non-zero means the code is + * incomplete; see usse_codegen_ex() opt_allow_unlowered. */ + unsigned nunlowered; + char unlowered[256]; +}; + +/* As uir_codegen(), plus the layout and an optional listing of the intended + * instructions (listing_cap entries). allow_unlowered turns a missing lowering + * from a hard error into a counted, reported omission. The literal pool is + * reported through res as well, so plain uir_codegen() callers also get it. */ +int usse_codegen_ex(const struct uir_shader *s, + const struct uir_codegen_opts *opts, + uint64_t *out, size_t out_cap, + struct uir_codegen_result *res, + struct usse_layout *lay, + struct usse_insn *listing, size_t listing_cap, + int allow_unlowered, + char *err, size_t errn); + +void usse_layout_free(struct usse_layout *lay); + + +/* The two IR passes the backend runs before it allocates registers, exposed + * so usse-irdiff can execute a shader before and after them. Both rewrite the + * shader in place and return how many instructions they removed or changed. */ +int usse_ir_copy_prop(struct uir_shader *s); +int usse_ir_dot_scalarise(struct uir_shader *s); + +#endif /* USSE_BACKEND_H */ diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/usse_cg.c mesa-26.2.2/src/gallium/drivers/sgx/usse_cg.c --- mesa-26.2.2.orig/src/gallium/drivers/sgx/usse_cg.c 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/usse_cg.c 2026-09-08 10:57:36.682386006 +0200 @@ -0,0 +1,5624 @@ +/* + * usse_cg.c - instruction selection, register allocation and code emission + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#include +#include +#include +#include + +#include "usse_backend.h" + +#define QUAD 4 /* hardware registers per IR value */ +#define NOREG 0xFFFFFFFFu +#define DEFAULT_TEMPS 64 +/* Scratch quads a lowering may name; see scratch_q(). */ +/* Four for the lowerings, and one more reserved for the write-after-read fix + * in emit_vec() - which runs inside a lowering and so must not take a quad + * that lowering is still holding. */ +#define SCRATCH_QUADS 5u +#define SCRATCH_WAR 4u +/* Registers a sampler's state block takes in the sa bank: {ctl, fmt, addr}. + * Must match SGX_SMP_SLOT in the driver. */ +#define SMP_SLOT 3u + +/* MOE per-slot request: increment 1, increment 0, or a swizzle. */ +#define MOE_INC1 0x0000 +#define MOE_INC0 0x0001 +#define MOE_SWZ(b) (0x100 | (b)) + +struct opnd { + uint8_t bank; + uint8_t base; + uint8_t swz; + uint8_t neg, abs; +}; + +struct cg { + const struct uir_shader *s; + struct uir_codegen_opts opts; + uint64_t *out; + size_t cap, n; + struct usse_insn *listing; + size_t lcap; + struct usse_layout *lay; + char *err; + size_t errn; + int failed, allow_unlowered, fragment, frag_out_packed; + int frag_out_rgba; + int frag_tex_preiterated, frag_tex_pa_base; + int vtx_uniform_pa_base, vtx_no_pool, vtx_limm, frag_in_bases; + int vtx_imm; /* materialise immediates as LIMM */ + uint32_t uni_temp; /* scratch quads the uniforms land in */ + uint32_t uni_live[UIR_MAX_UNIFORM]; /* where uniform i is, this insn */ + uint32_t imm_live[UIR_MAX_SRC]; /* where source k's immediate is */ + const struct uir_ref *cur_src; /* this instruction's source array */ + unsigned uni_used; /* scratch quads taken, this insn */ + struct uir_limm_ref limm[UIR_MAX_LIMM]; + uint32_t nlimm; + const unsigned char *vtx_out_slot; + const unsigned char *vtx_out_dw; + /* Which o[] slots the program writes. The MTE emits every dword up to + * the highest one, so a slot skipped between two written ones is + * still sent - as whatever the register held. */ + uint32_t out_slot_written; + + uint32_t *vbase, *vfirst, *vlast; + uint8_t *vwidth; + uint8_t *vused; /* components any def writes or any read selects */ + uint8_t *uni_reach; /* components of each uniform that are read */ + uint8_t *uni_umask; /* which components those reads actually select */ + uint8_t *uni_base; /* sa register each uniform starts at */ + uint32_t uni_cap; /* entries each of the three holds */ + uint32_t uni_regs; /* registers the uniforms take together */ + int relax_budget; /* measuring pass: do not enforce max_temps */ + uint32_t *scratch_at; /* scratch base per instruction */ + uint32_t *scratch_need; /* registers of scratch each one reaches */ + uint8_t *scratch_w; /* per instruction, per quad: width reached */ + uint32_t scratch_base[SCRATCH_QUADS]; /* this instruction's quads */ + uint8_t scratch_qwidth[SCRATCH_QUADS]; /* widths, this instruction */ + int scratch_placed; /* quads have bases of their own */ + uint32_t outreg_hole; /* fragment output stand-in, or NOREG */ + uint32_t *vorder; /* values by where their range starts */ + uint32_t fixed_top; /* first register above the whole-program areas */ + uint32_t nvirt; + + uint32_t nin, ntc, nuni, nout, nsamp; + uint32_t coord_f16; /* a sample took the packed-f16 coordinate */ + uint32_t tc_base; /* pa base of the first texcoord set */ + uint32_t smp_base; /* sa base of sampler 0's block */ + /* Quad of temporaries each packed input is unpacked into, or NOREG. */ + uint32_t in_unpack[16]; + uint32_t ntemps, npa, nsa; + + struct uir_literal pool[UIR_MAX_POOL]; + uint8_t pool_scalar[UIR_MAX_POOL]; /* quad holds shared scalars */ + uint8_t pool_used[UIR_MAX_POOL]; /* which components are taken */ + uint8_t pool_spill[UIR_MAX_POOL]; /* materialised, not banked */ + uint32_t pool_reads[UIR_MAX_POOL]; /* reads, to pick the spill */ + uint8_t last_op; /* what the previous slot holds */ + uint8_t in_war_fix; /* emit_vec() is already fixing one */ + uint32_t pairfix; /* padding inserted to split pairs */ + uint32_t npool, pool_base; /* literal pool, sa bank */ + uint32_t nkept; /* pool entries the bank holds */ + int imm_temp; /* a literal the bank cannot hold goes to a temp */ + uint32_t scratch; /* base of the scratch registers */ + uint32_t scratch_regs; /* how many are allocated */ + uint32_t scratch_used; /* most any one instruction took */ + uint8_t scratch_off[SCRATCH_QUADS]; /* logical quad -> offset */ + unsigned scratch_nreg; /* registers handed out, this insn */ + uint32_t fragout; /* stand-in for the fragment output */ + /* And for the second colour of a dual source blend, when the program + * declares one. Never written to the output bank: the blend reads it + * where it lies. */ + uint32_t fragout1; + int has_fragout1; + uint32_t killreg; /* discard condition, kept until EMIT */ + int uses_kill, killed; + + uint16_t moe[4]; + int moe_known; + /* What the next instruction emitted needs of the MOE. emit() puts the + * state right, so a caller that programs nothing gets the canonical + * increment-1 on every slot it steps. */ + uint16_t moe_req[4]; + int moe_req_set; + uint8_t *label_target; /* label ids a branch names */ + int eff; /* SGX_EFF_TRACE */ + int moe_old; /* SGX_MOE_OLD: the state churn back */ + + /* The EFO formation pass works on the finished instruction stream, so + * it needs every word back in decoded form together with the MOE + * state each one ran under. Kept only when the pass is on. */ + int efo; + struct usse_insn *slist; + uint16_t (*smoe)[4]; + size_t scap; + + uint32_t *labelpc; + uint32_t reg_values; /* registers the values themselves took */ + uint8_t last_end; + struct { uint32_t at, target; uint8_t pred; } *fix; + size_t nfix, fix_cap; +}; + +static void fail(struct cg *c, const char *fmt, ...) +{ + va_list ap; + + if (c->failed) + return; + c->failed = 1; + if (c->err && c->errn) { + va_start(ap, fmt); + vsnprintf(c->err, c->errn, fmt, ap); + va_end(ap); + } +} + +/* Register counts are what the driver builds its PDS data segment from, so + * they track what the code actually references, not what was reserved. */ +static void note_use(struct cg *c, const struct usse_operand *o) +{ + uint32_t top = (uint32_t)o->num + 1; + + switch (o->bank) { + case USSE_TEMP: if (top > c->ntemps) c->ntemps = top; break; + case USSE_PA: if (top > c->npa) c->npa = top; break; + case USSE_SA: if (top > c->nsa) c->nsa = top; break; + default: break; + } +} + +static unsigned chan(unsigned swz, unsigned k); +static uint32_t smp_slot_of(const struct cg *c, unsigned unit); +static uint32_t smp_slots_total(const struct cg *c); + +/* How far above its base register an operand reaches over `span` iterations. + * + * The MOE slot the instruction runs under decides the step, and adding + * span - 1 regardless assumed every slot was increment-1: an operand held at + * increment-0 was reported three registers too high, and one in swizzle mode + * was reported by its iteration count rather than by the channels the swizzle + * actually selects - "mov r0.xy, r4" under swizzle(zwzw) reads r6 and r7 and + * was counted as reaching r5. slot 0 is the destination, 1..3 the sources. */ +static unsigned moe_reach(const struct cg *c, unsigned slot, unsigned span) +{ + unsigned i, top = 0; + uint16_t m; + + if (!span) + return 0; + if (!c->moe_known || slot > 3) + return span - 1u; + m = c->moe[slot]; + if (m == MOE_INC0) + return 0; + if (!(m & 0x100)) + return span - 1u; + for (i = 0; i < span && i < 4; i++) + if (chan(m & 0xFFu, i) > top) + top = chan(m & 0xFFu, i); + return top; +} + +/* The register offset one MOE slot applies at iteration k. */ +static unsigned moe_off(uint16_t m, unsigned k) +{ + if (m == MOE_INC0) + return 0; + if (m & 0x100) + return chan(m & 0xFFu, k); + return k; +} + +/* Whether two slot programmings name the same registers over `span` + * iterations. An unused slot - span 0 - always does, which is what lets a + * state a previous instruction left stand for one that does not step it. + * + * The equivalence is exact, not a heuristic: increment-1 and swizzle(xyzw) + * both give offsets 0,1,2,3, and at one iteration every increment gives 0. + * A swizzle has four channels, so nothing beyond four iterations is + * comparable against one. */ +static int moe_slot_same(uint16_t a, uint16_t b, unsigned span) +{ + unsigned k; + + if (a == b) + return 1; + if (span > 4 && ((a | b) & 0x100)) + return 0; + for (k = 0; k < span; k++) + if (moe_off(a, k) != moe_off(b, k)) + return 0; + return 1; +} + +/* The cheapest MOE programming for a swizzled operand read over `span` + * iterations, and the register offset that goes with it. + * + * Swizzle mode is only needed where the channels are neither all the same nor + * consecutive. A replicate is increment 0 on the channel's own register, and + * a consecutive run is increment 1 on the first of them - and those two are + * the states every other instruction is already in, where a swizzle is a + * state nothing else shares. It is what made the four multiply-adds of a + * matrix product carry four SMLSIs where they need one. */ +static uint16_t moe_slot_for(const struct cg *c, unsigned swz, unsigned span, + unsigned *off) +{ + unsigned c0 = chan(swz, 0), k, rep = 1, run = 1; + + *off = 0; + if (c->moe_old) + return swz == UIR_SWZ_XYZW ? MOE_INC1 : MOE_SWZ(swz); + if (span > 4) + return MOE_SWZ(swz); + for (k = 1; k < span; k++) { + if (chan(swz, k) != c0) + rep = 0; + if (chan(swz, k) != c0 + k) + run = 0; + } + if (run) { + *off = c0; + return MOE_INC1; + } + if (rep) { + *off = c0; + return MOE_INC0; + } + return MOE_SWZ(swz); +} + +/* How many iterations an instruction runs. */ +static unsigned insn_span(const struct usse_insn *in) +{ + unsigned span; + + if (in->repeat) + span = in->repeat; + else /* mask mode stops after the highest enabled iteration */ + span = in->mask & 8 ? 4 : (in->mask & 4 ? 3 : + (in->mask & 2 ? 2 : (in->mask & 1 ? 1 : 0))); + /* An instruction with neither a repeat nor a mask still names its + * registers once. */ + if (!span && usse_op_has_dst(in->op)) + span = 1; + return span; +} + +/* How many iterations each MOE slot is stepped over: 0 where the instruction + * does not use the slot at all. + * + * A sample is the exception. Its coordinate operand names the first of two or + * three registers the unit reads for itself, and its state operand a block of + * them; whether the MOE steps those is not established, so a sample asks for + * the canonical state on every slot, which is what the code that emitted one + * straight used to program by hand. */ +static void insn_spans(const struct usse_insn *in, unsigned sp[4]) +{ + unsigned slots = usse_op_srcs(in->op), span = insn_span(in), i; + + if (in->op == USSE_SMP) { + for (i = 0; i < 4; i++) + sp[i] = 4; + return; + } + sp[0] = usse_op_has_dst(in->op) ? span : 0; + for (i = 0; i < 3; i++) + sp[i + 1] = ((slots >> i) & 1) ? span : 0; +} + +/* The instructions the part will not take two of in one pair. The full list + * also names LAPC, moves to and from the link register, PCOEFF and + * LOCK/RELEASE; this backend emits none of those, so a branch is all that can + * reach the rule here. */ +static int pair_ctrl(unsigned op) +{ + return op == USSE_BR || op == USSE_BA; +} + +static void emit(struct cg *c, struct usse_insn *in); +static void moe_reconcile(struct cg *c, const struct usse_insn *in); + +static void emit(struct cg *c, struct usse_insn *in) +{ + const char *e = NULL; + unsigned slots, i, span; + uint64_t w; + + if (c->failed) + return; + moe_reconcile(c, in); + if (c->failed) + return; + /* The USSE fetches instructions in pairs and refuses two branches in + * one. An instruction landing on an odd slot pairs with the one + * before it, so a NOP there splits them and the branch starts a pair + * of its own. Which pairs are illegal therefore depends on parity, + * and an unrelated instruction added anywhere earlier moves the + * breakage somewhere else - see work/win-gma500/. */ + if ((c->n & 1) && pair_ctrl(c->last_op) && pair_ctrl(in->op) && + !getenv("SGX_NO_PAIRFIX")) { + struct usse_insn pad; + uint32_t at0 = (uint32_t)c->n; + + memset(&pad, 0, sizeof pad); + pad.op = USSE_NOP; + emit(c, &pad); + /* A branch records its fixup before it is emitted, so the one + * just taken names the padding rather than the branch it was + * meant for. */ + for (i = 0; i < c->nfix; i++) + if (c->fix[i].at == at0) + c->fix[i].at = (uint32_t)c->n; + c->pairfix++; + } + /* Every instruction but a sample. .skipinv drops the instruction for + * the invalid pixels of a quad, so a sample carrying it never issues + * its dependent read for those lanes while the wdf that follows still + * waits on the counter - the task never resumes and the render stalls + * with no fault to explain it. The vendor's own output carries no + * .skipinv on a sample either. */ + if (in->op != USSE_SMP) + in->skipinv = 1; + /* Whether the program as emitted so far terminates. */ + c->last_end = in->end; + + /* An instruction with neither a repeat nor a mask still names its + * registers once. LIMM carries neither, so every literal a vertex + * program loads was counted as touching no register at all - the + * quad was reported only as far as whatever read it back happened to + * reach, up to three registers short. */ + slots = usse_op_srcs(in->op); + span = insn_span(in); + if (span) { + struct usse_operand o = in->dst; + + o.num = (uint8_t)(o.num + moe_reach(c, 0, span)); + note_use(c, &o); + note_use(c, &in->dst); + for (i = 0; i < 3; i++) { + if (!((slots >> i) & 1)) + continue; + note_use(c, &in->src[i]); + o = in->src[i]; + o.num = (uint8_t)(o.num + moe_reach(c, i + 1u, span)); + note_use(c, &o); + } + } + /* A sample's coordinate operand names its first register and the unit + * reads smp_dim of them, so counting the named one alone reported a + * volume's program as reading one primary attribute where it reads + * three. */ + if (in->op == USSE_SMP && in->smp_dim > 1) { + struct usse_operand o = in->src[0]; + + o.num = (uint8_t)(o.num + in->smp_dim - 1u); + note_use(c, &o); + } + if (usse_encode(in, &w, &e) != 0) { + /* The sa layout goes in the message: X takes the driver's + * stderr away, so the shader info log is the only channel. */ + { + char pm[128]; + unsigned k, o = 0, live = 0; + + for (k = 0; k < c->nuni; k++) { + unsigned b, m = c->uni_umask[k]; + + for (b = 0; b < 4; b++) + live += (m >> b) & 1u; + } + for (k = 0; k < c->npool && o + 4 < sizeof pm; k++) + o += (unsigned)snprintf(pm + o, sizeof pm - o, + "%x,", c->pool_used[k]); + pm[o ? o - 1 : 0] = 0; + fail(c, "instruction %zu: %s (%s) [sa: %u uniform(s) " + "reaching %u, live %u; %u sampler; pool from " + "%u, %u entries used %s]", c->n, + e ? e : "encode failed", usse_op_name(in->op), + c->nuni, c->uni_regs, live, c->nsamp, + c->pool_base, c->npool, pm); + } + return; + } + if (c->efo) { + if (c->n >= c->scap) { + size_t nc = c->scap ? c->scap * 2 : 256; + struct usse_insn *ni = realloc(c->slist, nc * sizeof *ni); + uint16_t (*nm)[4] = realloc(c->smoe, nc * sizeof *nm); + + if (ni) c->slist = ni; + if (nm) c->smoe = nm; + if (!ni || !nm) { + fail(c, "out of memory"); + return; + } + c->scap = nc; + } + c->slist[c->n] = *in; + memcpy(c->smoe[c->n], c->moe, sizeof c->moe); + } + /* SGX_EFF_TRACE: every register this instruction actually names, per + * iteration, with the MOE it runs under already applied - so an SMLSI + * itself prints nothing. + * + * This is what makes a change to the MOE provable rather than + * plausible. Reprogramming the state less often, or naming a register + * differently because the state differs, is sound exactly when this + * trace does not move; a change that alters which register an + * iteration reads shows up here as a diff, where the instruction + * counts and the encoder checks show nothing. */ + if (c->eff && in->op != USSE_SMLSI) { + unsigned k; + + fprintf(stderr, "%s m%x r%u", usse_op_name(in->op), + in->mask, in->repeat); + if (in->end) fprintf(stderr, " end"); + if (in->pred) fprintf(stderr, " p%u", in->pred); + for (k = 0; k < span; k++) { + if (usse_op_has_dst(in->op)) + fprintf(stderr, " |%u d=%u:%u", k, in->dst.bank, + in->dst.num + moe_off(c->moe[0], k)); + else + fprintf(stderr, " |%u", k); + for (i = 0; i < 3; i++) { + if (!((slots >> i) & 1)) + continue; + fprintf(stderr, " s%u=%u:%u%s%s", i, + in->src[i].bank, + in->src[i].num + + moe_off(c->moe[i + 1u], k), + in->src[i].neg ? "n" : "", + in->src[i].abs ? "a" : ""); + } + } + fputc('\n', stderr); + } + if (c->listing && c->n < c->lcap) + c->listing[c->n] = *in; + if (c->out && c->n < c->cap) + c->out[c->n] = w; + else if (c->out && c->n >= c->cap) { + fail(c, "output buffer holds %zu instructions, need more", c->cap); + return; + } + c->last_op = (uint8_t)in->op; + c->n++; +} + +/* Program the MOE. The state is written whole - one SMLSI carries all four + * slots - so this is the only place the recorded state changes. */ +static void emit_smlsi(struct cg *c, const uint16_t want[4]) +{ + struct usse_insn in; + unsigned i; + + memset(&in, 0, sizeof in); + in.op = USSE_SMLSI; + for (i = 0; i < 4; i++) { + if (want[i] & 0x100) { + in.moe_isswz[i] = 1; + in.moe_swz[i] = (uint8_t)(want[i] & 0xFF); + } else { + in.moe_inc[i] = (want[i] == MOE_INC0) ? 0 : 1; + } + } + memcpy(c->moe, want, 4 * sizeof want[0]); + c->moe_known = 1; + emit(c, &in); +} + +static const uint16_t moe_default[4] = { + MOE_INC1, MOE_INC1, MOE_INC1, MOE_INC1 +}; + +/* Whether the state in force redirects the registers of an instruction that + * runs once. Only a swizzle can: every increment contributes nothing at + * iteration 0. */ +static int moe_shifts_first(const uint16_t m[4]) +{ + unsigned i; + + for (i = 0; i < 4; i++) + if (moe_off(m[i], 0)) + return 1; + return 0; +} + +/* usc2's rule, groupinst.c:8561-8582: a state that shifts nothing on its + * first iteration cannot affect an unrepeated instruction, so a block with + * several predecessors can rely on that much even where the state itself is + * not known. Restoring it before a branch and before a join is what makes the + * state at a label worth something rather than nothing - without it every + * label cost an SMLSI, and this front end ends every shader with one. */ +static void moe_block_end(struct cg *c) +{ + if (c->moe_old) + return; + if (c->moe_known && moe_shifts_first(c->moe)) + emit_smlsi(c, moe_default); +} + +/* A join. The state is no longer exact, but moe_block_end() has established + * on every path here that it shifts no register at iteration 0, so the + * default stands in for it: instructions that run once may rely on it, and + * anything repeated reprograms the state outright. */ +static void moe_join(struct cg *c) +{ + if (!c->moe_old) + memcpy(c->moe, moe_default, sizeof c->moe); + c->moe_known = 0; +} + +/* Say what the next instruction emitted needs of the MOE. Nothing is emitted + * here: emit() decides, from what the instruction actually steps, whether the + * state in force already names the same registers. + * + * This used to program the state eagerly whenever it differed by a memcmp, + * which is what made 32% of this backend's output SMLSI: every scalar + * instruction asked for the canonical increments although at one iteration no + * increment applies, and every request named all four slots although most + * instructions step two. */ +static void moe_need(struct cg *c, const uint16_t want[4]) +{ + memcpy(c->moe_req, want, 4 * sizeof want[0]); + c->moe_req_set = 1; +} + +/* Put the state right for one instruction, if it is not already. + * + * A caller that programmed nothing gets increment 1 on every slot it steps, + * which is the canonical state and what every straight-emitted instruction in + * this backend used to ask for by hand. */ +static void moe_reconcile(struct cg *c, const struct usse_insn *in) +{ + uint16_t want[4]; + unsigned sp[4], i; + int need = 0, req = c->moe_req_set; + + c->moe_req_set = 0; + if (in->op == USSE_SMLSI) + return; + insn_spans(in, sp); + if (c->moe_old) + for (i = 0; i < 4; i++) + sp[i] = 4; + for (i = 0; i < 4; i++) + want[i] = req ? c->moe_req[i] : MOE_INC1; + for (i = 0; i < 4; i++) { + if (!sp[i]) + continue; + /* Where the state is not exact it is still known to shift + * nothing at iteration 0, which is all a slot stepped once + * asks of it. */ + if (!c->moe_known && sp[i] > 1) { + need = 1; + continue; + } + if (c->moe_old ? c->moe[i] != want[i] + : !moe_slot_same(c->moe[i], want[i], sp[i])) + need = 1; + } + if (!need) + return; + /* A slot this instruction does not step is left as it stands where + * that is known, so putting one slot right does not force the next + * instruction to put another back. */ + for (i = 0; i < 4; i++) + if (!sp[i]) + want[i] = c->moe_known ? c->moe[i] : MOE_INC1; + emit_smlsi(c, want); +} + +static unsigned popcount4(unsigned m) +{ + return ((m >> 0) & 1) + ((m >> 1) & 1) + ((m >> 2) & 1) + ((m >> 3) & 1); +} + +static unsigned chan(unsigned swz, unsigned k) +{ + return (swz >> (2 * k)) & 3; +} + +static unsigned mask_width(unsigned mask); +static struct opnd scratch_qw(struct cg *c, unsigned no, unsigned w); + +/* Emit one IR-level vector operation. slots is a bitmask of the USSE source + * slots used; a single-channel mask is resolved to plain scalar registers so + * no MOE programming is needed. */ +static void emit_vec(struct cg *c, unsigned uop, const struct opnd *d, + unsigned mask, const struct opnd s[3], unsigned slots, + const struct usse_insn *proto) +{ + struct usse_insn in = *proto; + uint16_t want[4] = { MOE_INC1, MOE_INC1, MOE_INC1, MOE_INC1 }; + unsigned i, span; + + if (!mask) + return; + in.op = (uint8_t)uop; + + /* MOV has no source negate/absolute; min(x,x) applies them and takes + * any bank, so it is the cheapest legal carrier. */ + if (uop == USSE_MOV && (s[1].neg || s[1].abs)) { + struct opnd s2[3]; + + memcpy(s2, s, sizeof s2); + s2[2] = s2[1]; + emit_vec(c, USSE_MIN, d, mask, s2, 0x6, proto); + return; + } + + if (popcount4(mask) == 1) { + unsigned k = mask & 1 ? 0 : (mask & 2 ? 1 : (mask & 4 ? 2 : 3)); + + in.mask = 1; + in.repeat = 0; + in.dst.bank = d->bank; + in.dst.num = (uint8_t)(d->base + k); + for (i = 0; i < 3; i++) { + if (!((slots >> i) & 1)) + continue; + in.src[i].bank = s[i].bank; + in.src[i].num = (uint8_t)(s[i].base + chan(s[i].swz, k)); + in.src[i].neg = s[i].neg; + in.src[i].abs = s[i].abs; + } + moe_need(c, want); + emit(c, &in); + return; + } + + /* A destination that shares a register with a source read through a + * swizzle is a write-after-read hazard. The MOE steps the destination + * one register an iteration while a swizzled source keeps reading the + * registers it started from, so the first iteration overwrites what + * the later ones still need - where the IR means all sources read + * before anything is written. + * + * It is what made a vec3 divided by a scalar come out shifted: + * `frcp r0, sa3` then `fmad r0.xyz, r0, sa0, sa4` computed (0,0,5) + * where (0,5,10) belongs, because iterations one and two read the r0 + * that iteration zero had already written. glmark2's ideas logo and + * its cast shadow rendered white for it - see + * work/hw-features/sgx_egl_vec3div.c. + * + * The result goes to a scratch quad of its own and is copied back. + * Only when the swizzle really differs: a destination and source + * stepping together read and write the same register each iteration, + * which is safe and much the commoner shape. */ + if (popcount4(mask) > 1 && !c->in_war_fix) { + unsigned hazard = 0, w = mask_width(mask); + + /* SGX_WAR_TRACE reports every vector emission this looks at, + * which is how a destination that had already been moved away + * from its source was told from one that had not. */ + if (getenv("SGX_WAR_TRACE")) + fprintf(stderr, "war: op %u dst bank %u base %u mask " + "%x slots %x | s0 bank %u base %u swz %x | " + "s1 %u/%u/%x | s2 %u/%u/%x\n", uop, d->bank, + d->base, mask, slots, s[0].bank, s[0].base, + s[0].swz, s[1].bank, s[1].base, s[1].swz, + s[2].bank, s[2].base, s[2].swz); + + for (i = 0; i < 3; i++) { + if (!((slots >> i) & 1)) + continue; + if (s[i].bank != d->bank || s[i].swz == UIR_SWZ_XYZW) + continue; + if (s[i].base < d->base + w && d->base < s[i].base + QUAD) + hazard = 1; + } + if (hazard) { + struct opnd t = scratch_qw(c, SCRATCH_WAR, w); + struct opnd mv[3]; + + c->in_war_fix = 1; + emit_vec(c, uop, &t, mask, s, slots, proto); + memset(mv, 0, sizeof mv); + mv[1] = t; + emit_vec(c, USSE_MOV, d, mask, mv, 0x2, proto); + c->in_war_fix = 0; + return; + } + } + + if (mask == 0xF) { + in.repeat = 4; + in.mask = 0; + } else { + in.repeat = 0; + in.mask = (uint8_t)mask; + } + span = insn_span(&in); + in.dst.bank = d->bank; + in.dst.num = (uint8_t)d->base; + for (i = 0; i < 3; i++) { + unsigned off; + + if (!((slots >> i) & 1)) + continue; + want[i + 1] = moe_slot_for(c, s[i].swz, span, &off); + in.src[i].bank = s[i].bank; + in.src[i].num = (uint8_t)(s[i].base + off); + in.src[i].neg = s[i].neg; + in.src[i].abs = s[i].abs; + } + moe_need(c, want); + emit(c, &in); +} + +static struct opnd konst_swz(unsigned num, unsigned swz, int neg) +{ + struct opnd o; + + memset(&o, 0, sizeof o); + o.bank = USSE_CONST; + o.base = (uint8_t)num; + o.swz = (uint8_t)swz; + o.neg = (uint8_t)(neg != 0); + return o; +} + +static struct opnd konst(unsigned num, int neg) +{ + return konst_swz(num, UIR_SWZ_XYZW, neg); +} + +static struct opnd temp(unsigned base) +{ + struct opnd o; + + memset(&o, 0, sizeof o); + o.bank = USSE_TEMP; + o.base = (uint8_t)base; + o.swz = UIR_SWZ_XYZW; + return o; +} + +/* A scratch quad. No scratch value outlives the instruction that took it - + * that is why a discard keeps its condition in killreg instead - so the quads + * a lowering names are handed out per instruction and reused by the next one. + * A sample names quads two and three and so used to reserve four; it takes + * two. What a shader pays is the most any single instruction asked for. */ +/* Registers a copy of this mask occupies: the channels above it are left + * undefined by the copy itself, so nothing may read them. */ +static unsigned mask_width(unsigned mask) +{ + if (!mask) + return QUAD; + return mask & 8 ? 4 : mask & 4 ? 3 : mask & 2 ? 2 : 1; +} + +static struct opnd scratch_qw(struct cg *c, unsigned no, unsigned w) +{ + static int fixed = -1; + + if (no >= SCRATCH_QUADS) { + fail(c, "scratch quad %u out of range", no); + no = 0; + } + if (!w || w > QUAD) + w = QUAD; + /* SGX_SCRATCH_FIXED restores the quad-per-name layout this replaced, + * so the two can be compared on hardware. */ + if (fixed < 0) + fixed = getenv("SGX_SCRATCH_FIXED") != NULL; + if (fixed) { + if (QUAD * (no + 1u) > c->scratch_used) + c->scratch_used = QUAD * (no + 1u); + return temp(c->scratch + QUAD * no); + } + /* Each logical quad has a base of its own, so they can be put in + * separate holes among the values. Packed into one block they needed a + * single free run as wide as every quad together - twelve registers for + * alacritty's fragment program, which never fits - where four apiece + * usually does. */ + if (w > c->scratch_qwidth[no]) + c->scratch_qwidth[no] = (uint8_t)w; + if (c->scratch_off[no] == 0xff) { + c->scratch_off[no] = (uint8_t)c->scratch_nreg; + c->scratch_nreg += w; + if (c->scratch_nreg > c->scratch_used) + c->scratch_used = c->scratch_nreg; + } + if (c->scratch_placed) + return temp(c->scratch_base[no]); + return temp(c->scratch + c->scratch_off[no]); +} + +static struct opnd scratch_q(struct cg *c, unsigned no) +{ + return scratch_qw(c, no, QUAD); +} + +/* Every instruction starts with all four quads unclaimed. */ +static void scratch_reset(struct cg *c) +{ + memset(c->scratch_off, 0xff, sizeof c->scratch_off); + memset(c->scratch_qwidth, 0, sizeof c->scratch_qwidth); + c->scratch_nreg = 0; +} + +/* src0 of the MAD family and the MOVC condition can only name the temporary + * or primary attribute bank; anything else is copied into a scratch quad. */ +static struct opnd legal_src0(struct cg *c, struct opnd v, unsigned scratch_no, + unsigned mask) +{ + struct usse_insn proto; + struct opnd s[3], t; + + if (v.bank == USSE_TEMP || v.bank == USSE_PA) + return v; + memset(&proto, 0, sizeof proto); + memset(s, 0, sizeof s); + t = scratch_qw(c, scratch_no, mask_width(mask)); + s[1] = v; + emit_vec(c, USSE_MOV, &t, mask ? mask : 0xF, s, 0x2, &proto); + return t; +} + +/* Copy a value into scratch so that forms without source modifiers, or that + * need consecutive registers, can use it. */ +static struct opnd flatten_w(struct cg *c, struct opnd v, unsigned scratch_no, + unsigned mask, unsigned w); + +static struct opnd flatten(struct cg *c, struct opnd v, unsigned scratch_no, + unsigned mask) +{ + return flatten_w(c, v, scratch_no, mask, mask_width(mask)); +} + +static struct opnd flatten_w(struct cg *c, struct opnd v, unsigned scratch_no, + unsigned mask, unsigned w) +{ + struct usse_insn proto; + struct opnd s[3], t; + + if (!v.neg && !v.abs && v.swz == UIR_SWZ_XYZW) + return v; + memset(&proto, 0, sizeof proto); + memset(s, 0, sizeof s); + t = scratch_qw(c, scratch_no, w); + s[1] = v; + emit_vec(c, USSE_MOV, &t, mask ? mask : 0xF, s, 0x2, &proto); + return t; +} + +/* --- register layout --------------------------------------------------- */ + +static void note_map(struct cg *c, enum uir_regclass cls, uint32_t idx, uint32_t base) +{ + struct usse_map_entry *m; + + if (!c->lay) + return; + if (c->lay->nmap == c->lay->map_cap) { + size_t nc = c->lay->map_cap ? c->lay->map_cap * 2 : 16; + m = realloc(c->lay->map, nc * sizeof *m); + if (!m) + return; + c->lay->map = m; + c->lay->map_cap = nc; + } + m = &c->lay->map[c->lay->nmap++]; + m->cls = cls; + m->index = idx; + m->base = base; + m->count = QUAD; +} + +/* Interns one vec4 literal and returns the sa register it starts at. Entries + * are deduplicated on the exact bit pattern, so -0.0 and 0.0 stay distinct. */ +/* Interns one vec4 literal and returns the sa register it starts at, with the + * component to read it from in *comp - four when the whole quad is the value. + * + * A literal whose four channels are equal is a scalar, and a scalar can be read + * from any component, so four of them share one quad and the reader is given a + * swizzle that picks its own. The bank holds thirty-two registers and a quad + * apiece is what put alacritty's fragment program past it: seven uniforms and + * two literal quads came to thirty-six. */ +/* The two values the hardware holds outright, so a literal equal to one of + * them costs no bank register at all. */ +/* The hardware constant bank, sgxdefs.h:7653. Only the values that are exact + * in binary floating point are listed: the bank also holds e, 1/e, sqrt2, + * 1/sqrt2, the pi family at 32-38 and Taylor coefficients at 40-44, but which + * f32 the silicon holds for those is not established, and folding a literal + * onto a value that differs in the last bit would be silently wrong. + * + * 48-51 are four zeroes and 52-55 four ones, which is why those two alone can + * be read as a whole quad; every other entry is a single register. */ +static const struct { uint8_t num; float v; } hw_const[] = { + { 3, 1.0f }, { 4, 2.0f }, { 5, 4.0f }, + { 6, 8.0f }, { 7, 16.0f }, { 8, 32.0f }, + { 9, 64.0f }, { 10, 128.0f }, { 11, 256.0f }, + { 12, 512.0f }, { 13, 1024.0f }, + { 14, 0.5f }, { 15, 0.25f }, { 16, 0.125f }, + { 17, 0.0625f }, { 18, 0.03125f }, { 19, 0.015625f }, + { 20, 1.0f / 128.0f }, { 21, 1.0f / 256.0f }, + { 22, 1.0f / 512.0f }, { 23, 1.0f / 1024.0f }, + { 24, 1.0f / 2048.0f }, { 25, 1.0f / 4096.0f }, + { 26, 1.0f / 8192.0f }, { 27, 1.0f / 16384.0f }, + { 39, 1.0f / 65536.0f }, +}; + +/* Whether the bank is used past zero and one. Off: see fold_hw_const(). */ +static int hwconst_wanted(void) +{ + static int on = -1; + + if (on < 0) + on = getenv("SGX_HWCONST") != NULL; + return on; +} + +static int hw_const_scalar(float f, unsigned *num) +{ + unsigned k; + + for (k = 0; k < sizeof hw_const / sizeof hw_const[0]; k++) + if (hw_const[k].v == f) { + *num = hw_const[k].num; + return 1; + } + return 0; +} + +/* Whether this *read* of a literal can come from the bank, which depends on + * the swizzle as much as on the value. The front end packs unrelated scalars + * into one vec4 - Mesa hands us #(0.25,0.125,0.0625,1) and reads .zzzz of it - + * so the question is not "is this quad a bank value" but "does this read + * select one component that is". A replicated swizzle selects one; anything + * else needs the quad in the bank, and an entry's neighbours are unrelated + * values. The two runs of four - 48-51 zeroes, 52-55 ones - are the exception + * and can be read whole. + * + * Returns the register and the swizzle to read it with; the operand's register + * number is base plus the swizzle's channel, see emit_vec(). */ +static int imm_hw_const(const struct uir_ref *r, int fragment, unsigned *num, + unsigned *swz) +{ + float e[4]; + int one = 1, zero = 1, same = 1; + unsigned k; + + /* The swizzle picks the channel, abs applies to it and the negate + * comes last - so work out what each channel of this read is actually + * worth and look *that* up. Folding the value and dropping the + * modifiers reads -1.0 as +1.0, which is how varyuni turned black. */ + for (k = 0; k < 4; k++) { + float f = r->imm[chan(r->swizzle, k)]; + + if (r->absolute) + f = f < 0.0f ? -f : f; + if (r->negate) + f = -f; + e[k] = f; + if (f != 1.0f) + one = 0; + if (f != 0.0f) + zero = 0; + if (f != e[0]) + same = 0; + } + /* Replicated, so every channel reads the entry the DDK names first. + * The bank holds each of these values four times over - 48..51 zero, + * 52..55 one, ZERO/ZERO4..6 and FLOAT1/FLOAT1_2..4 - and reading the + * run as a quad picks a different entry per channel: a literal in the + * w channel took base + 3, entry 55. + * + * Entry 55 stops the render the same way entry 14 does. A vertex + * program that writes vec4(d, d, d, 1.0) reads it for the alpha and + * loses every frame, while the same program writing a two-component + * varying does not - identical instruction count, identical temporary + * count, and c55 the only difference in the emitted code. + * + * The duplicates buy nothing, so none of them are used. */ + /* A fragment program reads none of the bank. Entry 14 stops the render + * outright and so does entry 52, the DDK's FLOAT1 - the canonical one, + * which this used to keep on the grounds that it was proven. It is + * not: the same shader taking its literal 1.0 from a secondary + * attribute instead runs at 21 ms where the bank read loses every + * frame, and that one instruction is the only difference between the + * two programs. + * + * The vertex stage is not affected - its own programs read 48 and 52 + * and retire - so this is a fragment-side restriction rather than a + * bad entry. SGX_HWCONST=1 puts the fold back. */ + if (fragment && !hwconst_wanted()) + return 0; + *swz = UIR_SWZ_XXXX; + if (one) { + *num = USSE_C_ONE; + return 1; + } + if (zero) { + *num = USSE_C_ZERO; + return 1; + } + /* Zero and one are folded above and are proven on both stages. The + * rest of the bank is not: a fragment program reading entry 14, the + * DDK's FLOAT1OVER2, never retires. The core makes no progress at + * all, every clock gate stays open, and the kernel times the render + * out as a locked core. + * + * That one instruction is the whole of it. Diffing the emitted code + * for a shader that stalls against the same shader with this fold + * off leaves a single line - `mov r2, c14` against `mov r2, sa0` - + * and the frame time goes from 177 ms to 30. It is what held every + * glmark2 scene with a varying wider than two components down, build + * and shading among them, because those are the ones that need a + * literal the narrow shaders happened to take from a uniform. + * + * An earlier note here recorded the opposite, that a fragment program + * reads 0.5 from entry 14 correctly. Whatever that measured, it does + * not hold for the shapes glmark2 draws. + * + * So the bank is used for the two values that are proven and the rest + * comes from the secondary attributes, which costs a register and + * works. SGX_HWCONST=1 puts the wider fold back for whoever wants to + * establish which entries are safe; it is not a per-entry blocklist + * because nothing has established that 14 is the only bad one. */ + if (!hwconst_wanted()) + return 0; + if (!fragment) + return 0; + /* Anything else has to be one value in every channel: an entry's + * neighbours in the bank are unrelated, so only a replicated read can + * come from it. */ + if (!same || !hw_const_scalar(e[0], num)) + return 0; + *swz = UIR_SWZ_XXXX; + return 1; +} + +/* What the uniforms take in the sa bank. Wanted twice: to lay them out, and + * before that to tell whether the literal pool will still fit behind them. */ +static unsigned uni_span(const struct cg *c) +{ + unsigned i, b = 0, nopack = getenv("SGX_NO_UNI_PACK") != NULL; + + if (!c->fragment && c->vtx_uniform_pa_base) + return 0; + for (i = 0; i < c->nuni; i++) + b += (!nopack && c->uni_reach[i]) ? c->uni_reach[i] : + (unsigned)QUAD; + return b; +} + +/* Returns the entry index, not the register: a spilled entry has no register. + * Callers that want one read pool[index].sa_base. */ +static uint32_t pool_add_c(struct cg *c, const float v[4], unsigned *comp) +{ + struct uir_literal *e; + uint32_t i, k; + int scalar = v[0] == v[1] && v[1] == v[2] && v[2] == v[3]; + + *comp = 4; + for (i = 0; i < c->npool; i++) + if (!memcmp(c->pool[i].v, v, 4 * sizeof(float))) + return i; + if (scalar) + for (i = 0; i < c->npool; i++) { + if (!c->pool_scalar[i]) + continue; + for (k = 0; k < 4; k++) + if (c->pool_used[i] & (1u << k)) { + if (c->pool[i].v[k] == v[0]) { + *comp = k; + return i; + } + } else { + c->pool[i].v[k] = v[0]; + memcpy(&c->pool[i].bits[k], + &c->pool[i].v[k], + sizeof c->pool[i].bits[k]); + c->pool_used[i] |= 1u << k; + *comp = k; + return i; + } + } + if (c->npool == UIR_MAX_POOL) { + fail(c, "constant pool full (%d vec4 entries)", UIR_MAX_POOL); + return 0; + } + e = &c->pool[c->npool]; + memcpy(e->v, v, 4 * sizeof(float)); + for (k = 0; k < 4; k++) + memcpy(&e->bits[k], &e->v[k], sizeof e->bits[k]); + e->sa_base = c->pool_base + QUAD * c->nkept++; + e->sa_count = QUAD; + c->pool_scalar[c->npool] = (uint8_t)scalar; + c->pool_used[c->npool] = scalar ? 1u : 0xfu; + if (scalar) + *comp = 0; + return c->npool++; +} + + + +static struct opnd resolve(struct cg *c, const struct uir_ref *r) +{ + struct opnd o; + + memset(&o, 0, sizeof o); + o.swz = r->swizzle; + o.neg = r->negate; + o.abs = r->absolute; + + switch (r->cls) { + case UIR_REG_VIRT: + if (r->index >= c->nvirt || c->vbase[r->index] == NOREG) { + fail(c, "virtual register v%u was never defined", r->index); + o.bank = USSE_TEMP; + return o; + } + o.bank = USSE_TEMP; + o.base = (uint8_t)c->vbase[r->index]; + return o; + case UIR_REG_IN: + /* Unpacked at entry, so the program reads floats. */ + if (r->index < 16 && c->in_unpack[r->index] != NOREG) { + o.bank = USSE_TEMP; + o.base = (uint8_t)c->in_unpack[r->index]; + return o; + } + o.bank = USSE_PA; + o.base = (uint8_t)(c->frag_in_bases && r->index < 16 ? + c->opts.frag_in_base[r->index] : + QUAD * r->index); + return o; + case UIR_REG_TEXCOORD: + o.bank = USSE_PA; + o.base = (uint8_t)(c->tc_base + QUAD * r->index); + return o; + case UIR_REG_UNIFORM: + /* Materialised into temporaries by LIMM, so they cost no + * attribute register at all. */ + if (!c->fragment && c->vtx_limm) { + if (r->index >= UIR_MAX_UNIFORM || + c->uni_live[r->index] == NOREG) { + fail(c, "uniform %u was not materialised", + r->index); + o.bank = USSE_TEMP; + return o; + } + o.bank = USSE_TEMP; + o.base = (uint8_t)c->uni_live[r->index]; + return o; + } + /* Non-zero says the uniforms were placed in the primary bank + * at that offset; zero leaves them in the secondary one. */ + if (!c->fragment && c->vtx_uniform_pa_base) { + o.bank = USSE_PA; + o.base = (uint8_t)(c->vtx_uniform_pa_base + + QUAD * r->index); + return o; + } + o.bank = USSE_SA; + /* Where the layout pass actually put it, not a quad per + * uniform. The two disagreed: alloc_regs() packs each uniform + * into the components it is read through and reports those + * bases to the driver, which fills the bank to match - while + * this read sa[4 * i] regardless. A program with two uniforms + * therefore got the first, because 4 * 0 is 0, and read the + * second out of a register nothing had written. */ + o.base = (uint8_t)((c->uni_base && r->index < c->nuni) ? + c->uni_base[r->index] : + QUAD * r->index); + return o; + case UIR_REG_OUT: + if (c->fragment) { + o.bank = USSE_TEMP; + /* The index decides which colour: a dual source + * program writes its second into its own quad, and + * the two used to share one - both outputs landing on + * the same registers. */ + o.base = (uint8_t)((r->index && c->has_fragout1) ? + c->fragout1 : c->fragout); + } else { + unsigned slot = r->index; + + if (c->vtx_out_slot && slot < 16) + slot = c->vtx_out_slot[slot]; + o.bank = USSE_OUT; + /* The packed offset when the caller supplied one, + * so a set narrower than four floats does not leave + * the sets behind it two dwords out from where the + * iterator reads them. */ + o.base = (uint8_t)((c->vtx_out_dw && r->index < 16) ? + c->vtx_out_dw[r->index] : + QUAD * slot); + if (slot < 32) + c->out_slot_written |= 1u << slot; + } + return o; + case UIR_REG_IMM: + /* A vertex program cannot reach the literal pool: the pool + * lives in the secondary attribute bank, and this stage has + * no secondary PDS program to fill it. The two values the + * hardware holds outright cover what a vertex program + * actually needs - the 1.0 a position's w is set to and the + * 0.0 a multiply is lowered against - and anything else has + * to be refused rather than read as whatever the bank held. */ + /* Materialised in front of the instruction that reads it, + * which resolve() finds by where the operand sits in that + * instruction's source array. */ + if ((c->vtx_imm || c->imm_temp) && c->cur_src && + r >= c->cur_src && r < c->cur_src + UIR_MAX_SRC) { + size_t k = (size_t)(r - c->cur_src); + + if (c->imm_live[k] != NOREG) { + o.bank = USSE_TEMP; + o.base = (uint8_t)c->imm_live[k]; + return o; + } + } + /* The two values the hardware holds outright cost no bank + * register, so fold them in every program rather than only in + * a vertex one that has no pool to fall back on: the X + * server's composite needed 130 registers of a 128-register + * bank, and its literal zeroes and ones were 4 of them. */ + { + unsigned hw, hwswz; + + /* The modifiers are already in the value chosen, so + * the operand carries neither. */ + if (imm_hw_const(r, c->fragment, &hw, &hwswz)) + return konst_swz(hw, hwswz, 0); + } + if (!c->fragment && c->vtx_no_pool) { + fail(c, "a vertex program has no literal pool, so the " + "constant %g cannot be reached", (double)r->imm[0]); + o.bank = USSE_TEMP; + return o; + } + { + unsigned comp; + + o.bank = USSE_SA; + o.base = (uint8_t)c->pool[pool_add_c(c, r->imm, + &comp)].sa_base; + /* A scalar sharing a quad is read from its own + * component; every channel of it held the same value + * before, so replacing the swizzle changes nothing + * except where it is read from. */ + if (comp < 4) + o.swz = (uint8_t)UIR_SWZ(comp, comp, comp, + comp); + return o; + } + default: + fail(c, "register class %d cannot be used as a value", (int)r->cls); + o.bank = USSE_TEMP; + return o; + } +} + +/* An operation that works channel by channel reads a source channel only + * where it writes one, so its destination mask bounds what the source has to + * hold. The rest - dot products, normalise, a sample coordinate, a discard or + * a branch condition - consume the whole vector whatever they write. */ +static int elementwise(enum uir_op op) +{ + switch (op) { + case UIR_MOV: case UIR_ADD: case UIR_SUB: case UIR_MUL: case UIR_MAD: + case UIR_DDX: case UIR_DDY: + case UIR_MIN: case UIR_MAX: case UIR_FRC: case UIR_FLOOR: + case UIR_RCP: case UIR_RSQ: case UIR_LOG2: case UIR_EXP2: + case UIR_SQRT: case UIR_SLT: case UIR_SGE: case UIR_SEQ: + case UIR_SNE: case UIR_CMP: + return 1; + default: + return 0; + } +} + +/* Widest channel any read of this reference can touch, bounded by the + * channels the reading instruction actually produces. Counting all four of an + * identity swizzle gave every value a whole quad and put shaders that branch + * past the temporary count the hardware can name. */ +static unsigned swz_reach(const struct uir_ref *r, unsigned mask) +{ + unsigned i, m = 0; + + for (i = 0; i < 4; i++) + if ((mask >> i) & 1 && chan(r->swizzle, i) + 1 > m) + m = chan(r->swizzle, i) + 1; + return m; +} + +/* Which source components the reads select, as against how far they reach: a + * uniform read only at .w reaches four registers but uses one. */ +static unsigned swz_mask(const struct uir_ref *r, unsigned mask) +{ + unsigned i, m = 0; + + for (i = 0; i < 4; i++) + if ((mask >> i) & 1) + m |= 1u << chan(r->swizzle, i); + return m; +} + +/* --- live-range splitting ---------------------------------------------- */ + +/* The front end numbers a value once and reuses it: one TGSI temporary holds + * unrelated quantities at different points of the program. Liveness then fuses + * them into a single value running from the first definition to the last use, + * as wide as the widest use - so a scalar sharing a number with a vector costs + * four registers for the length of the shader. Splitting each number into its + * def-use webs gives the allocator the ranges the program really has. + * + * A web is the transitive closure of "this use reads that definition", taken + * per component: a partial write must not appear to kill the definition that + * supplies the other channels. + * + * Measured on the X server's glyph composite, the shader this was written for: + * thirteen values reaching thirty-two registers become fifty-nine webs + * reaching twenty-seven. */ + +#define SPLIT_MAX_WORDS (16u * 1024u * 1024u) /* reaching-set budget */ +#define SPLIT_MAX_ITERS 4096u + +static uint32_t uf_find(uint32_t *p, uint32_t x) +{ + while (p[x] != x) { + p[x] = p[p[x]]; + x = p[x]; + } + return x; +} + +static void uf_union(uint32_t *p, uint32_t a, uint32_t b) +{ + a = uf_find(p, a); + b = uf_find(p, b); + if (a != b) + p[a] = b; +} + +/* Which components of a source an instruction reads, matching what scan() + * charges the value for. */ +static unsigned src_rmask(const struct uir_insn *n, unsigned j) +{ + return elementwise(n->op) ? n->writemask : + (n->op == UIR_TEX && j == 0) ? 0x3u : 0xfu; +} + +static unsigned insn_succs(const struct uir_shader *s, const uint32_t *lbl, + size_t i, size_t *sv) +{ + const struct uir_insn *n = &s->insns[i]; + unsigned k = 0; + + if (n->op == UIR_BR) { + if (n->branch_target <= s->next_label && + lbl[n->branch_target] != NOREG) + sv[k++] = lbl[n->branch_target]; + return k; + } + if (i + 1 < s->ninsns) + sv[k++] = i + 1; + if (n->op == UIR_BRC && n->branch_target <= s->next_label && + lbl[n->branch_target] != NOREG) + sv[k++] = lbl[n->branch_target]; + return k; +} + +/* Which IR operations write their destination. An instruction that does not + * still carries a zeroed dst, which reads as virtual register 0 - counting + * those as writes made every value 0 look many-times-written and stopped both + * passes below from firing at all. */ +static int ir_has_dst(enum uir_op op) +{ + switch (op) { + case UIR_NOP: case UIR_KILL: case UIR_EMIT: + case UIR_BR: case UIR_BRC: case UIR_LABEL: case UIR_RET: + return 0; + default: + return 1; + } +} + +/* Compose two swizzles: read `outer` through `inner`. Both are two bits a + * channel, x in the low pair. */ +static uint8_t swz_compose(uint8_t inner, uint8_t outer) +{ + unsigned k, r = 0; + + for (k = 0; k < 4; k++) + r |= chan(inner, chan(outer, k)) << (2 * k); + return (uint8_t)r; +} + +/* Copy propagation. + * + * `mov vD, vS.swz` where vD is a virtual written exactly once and read only + * further down the same straight-line run, and vS is not written again before + * the last of those reads, is a rename: every read of vD becomes a read of vS + * through the composed swizzle, and the move goes. 391 of the 1814 + * instructions this backend emitted for its own corpus were moves. + * + * Confined to one run of instructions on purpose. A value that crosses a + * branch or a label needs a real liveness analysis to propagate, and getting + * that wrong on a branchy shader is exactly what live-range splitting in this + * file already does - which is why that one is behind SGX_SPLIT. A run ends + * at a label and at a branch, so no path but the fall-through can reach the + * uses this rewrites. + * + * Refuses, and each refusal is a case that would be wrong: + * - a partial write, which leaves the other channels of vD to some other + * instruction; + * - a saturating move, or one with a modifier the reader would have to + * carry; + * - a second write of vD anywhere, so the read that follows may not be of + * this value; + * - a write of vS between the move and the last read of vD, which is the + * write the copy existed to get out of the way of; + * - a destination or a source that is not a virtual: an output, an input + * or a uniform is placed by the hardware ABI, not by the allocator. + */ +int usse_ir_copy_prop(struct uir_shader *s) +{ + size_t n = s->ninsns, i, j; + uint32_t *ndef; + int removed = 0, again = 1; + unsigned pass; + + if (!n || !s->nvirt) + return 0; + ndef = calloc(s->nvirt, sizeof *ndef); + if (!ndef) + return 0; + for (i = 0; i < n; i++) + if (ir_has_dst(s->insns[i].op) && + s->insns[i].dst.cls == UIR_REG_VIRT && + s->insns[i].dst.index < s->nvirt) + ndef[s->insns[i].dst.index]++; + + /* Deleting one copy can expose the next in a chain, so this runs to a + * fixed point. The bound is a bound, not an expectation. */ + for (pass = 0; again && pass < 8; pass++) { + again = 0; + for (i = 0; i < n; i++) { + struct uir_insn *m = &s->insns[i]; + size_t last = i, end; + uint32_t D, S; + int ok = 1; + + if (m->op != UIR_MOV || m->saturate || + m->writemask != 0xF) + continue; + if (m->dst.cls != UIR_REG_VIRT || + m->src[0].cls != UIR_REG_VIRT) + continue; + if (m->src[0].negate || m->src[0].absolute) + continue; + D = m->dst.index; + S = m->src[0].index; + if (D >= s->nvirt || S >= s->nvirt || D == S) + continue; + if (ndef[D] != 1) + continue; + + /* The run this move sits in. */ + for (end = i + 1; end < n; end++) { + enum uir_op o = s->insns[end].op; + + if (o == UIR_LABEL) + break; + if (o == UIR_BR || o == UIR_BRC) { + end++; + break; + } + } + + /* Every read of vD must be inside it, and nothing may + * write vS before the last of them. */ + for (j = 0; j < n && ok; j++) { + const struct uir_insn *u = &s->insns[j]; + unsigned k; + + if (j == i) + continue; + for (k = 0; k < u->nsrc; k++) { + if (u->src[k].cls != UIR_REG_VIRT || + u->src[k].index != D) + continue; + if (j < i || j >= end) + ok = 0; + else if (j > last) + last = j; + } + } + if (!ok || last == i) + continue; + for (j = i + 1; j <= last; j++) + if (ir_has_dst(s->insns[j].op) && + s->insns[j].dst.cls == UIR_REG_VIRT && + s->insns[j].dst.index == S) + ok = 0; + if (!ok) + continue; + + for (j = i + 1; j <= last; j++) { + struct uir_insn *u = &s->insns[j]; + unsigned k; + + for (k = 0; k < u->nsrc; k++) { + if (u->src[k].cls != UIR_REG_VIRT || + u->src[k].index != D) + continue; + u->src[k].index = S; + u->src[k].swizzle = + swz_compose(m->src[0].swizzle, + u->src[k].swizzle); + } + } + ndef[D] = 0; + memset(m, 0, sizeof *m); + m->op = UIR_NOP; + m->dst.index = s->nvirt; /* named by nothing */ + removed++; + again = 1; + } + } + free(ndef); + /* Compacted rather than left as NOPs: the scan that follows records a + * value's first and last instruction, and a dead one still standing + * where the copy was would hold a live range open across it. */ + if (removed) { + size_t k = 0; + + for (i = 0; i < n; i++) + if (s->insns[i].op != UIR_NOP) + s->insns[k++] = s->insns[i]; + s->ninsns = k; + } + return removed; +} + +/* A dot product broadcast into a quad, made scalar. + * + * DP3 and DP4 replicate one number across the destination mask, so the + * backend accumulated into a scratch register and then copied it out four + * times - `fdp.repeat3 r8` then `mov.repeat4 r0, r8`. Every channel of the + * result is the same number, so a reader can take it from one register + * through .xxxx and the copy goes with the other three registers. It is what + * the vendor's own output does: the dot lands in one place and everything + * reads it there. + * + * Sound because the value is a broadcast: any swizzle of it selects the same + * number, so rewriting every read to .xxxx cannot change what is read. The + * conditions are that nothing else writes the value - a second write would + * fill channels this no longer writes - and that every read comes after the + * write, so no read is of the wider form. */ +int usse_ir_dot_scalarise(struct uir_shader *s) +{ + size_t n = s->ninsns, i, j; + uint32_t *ndef; + int done = 0; + + if (!n || !s->nvirt) + return 0; + ndef = calloc(s->nvirt, sizeof *ndef); + if (!ndef) + return 0; + for (i = 0; i < n; i++) + if (ir_has_dst(s->insns[i].op) && + s->insns[i].dst.cls == UIR_REG_VIRT && + s->insns[i].dst.index < s->nvirt) + ndef[s->insns[i].dst.index]++; + + for (i = 0; i < n; i++) { + struct uir_insn *d = &s->insns[i]; + uint32_t D; + int ok = 1; + + if ((d->op != UIR_DP3 && d->op != UIR_DP4) || d->saturate) + continue; + if (d->dst.cls != UIR_REG_VIRT || d->dst.index >= s->nvirt) + continue; + if (popcount4(d->writemask) < 2) + continue; + D = d->dst.index; + if (ndef[D] != 1) + continue; + for (j = 0; j < n && ok; j++) { + const struct uir_insn *u = &s->insns[j]; + unsigned k; + + if (j == i) + continue; + for (k = 0; k < u->nsrc; k++) + if (u->src[k].cls == UIR_REG_VIRT && + u->src[k].index == D && j < i) + ok = 0; + } + if (!ok) + continue; + for (j = i + 1; j < n; j++) { + struct uir_insn *u = &s->insns[j]; + unsigned k; + + for (k = 0; k < u->nsrc; k++) + if (u->src[k].cls == UIR_REG_VIRT && + u->src[k].index == D) + u->src[k].swizzle = UIR_SWZ_XXXX; + } + d->writemask = 1; + done++; + } + free(ndef); + return done; +} + +/* Returns a renumbered copy, or NULL to compile the shader as it stands - + * splitting is an optimisation, so running out of room for the analysis or + * failing to converge costs registers, not correctness. */ +static struct uir_shader *split_ranges(const struct uir_shader *s) +{ + size_t n = s->ninsns, i; + uint32_t nv = s->nvirt, ndef = 0, bw, nnew = 0, iters = 0; + uint32_t *defid = NULL, *parent = NULL, *in = NULL, *tmp = NULL; + uint32_t *lbl = NULL, *newidx = NULL; + struct uir_insn *ins = NULL; + struct uir_shader *out = NULL; + uint64_t need; + unsigned j, k; + int changed; + + if (!nv || !n) + return NULL; + + defid = malloc(n * sizeof *defid); + lbl = malloc(((size_t)s->next_label + 1) * sizeof *lbl); + if (!defid || !lbl) + goto done; + for (i = 0; i <= s->next_label; i++) + lbl[i] = NOREG; + for (i = 0; i < n; i++) { + const struct uir_insn *ni = &s->insns[i]; + + if (ni->op == UIR_LABEL && ni->label_id <= s->next_label) + lbl[ni->label_id] = (uint32_t)i; + defid[i] = (ni->dst.cls == UIR_REG_VIRT && ni->dst.index < nv) ? + ndef++ : NOREG; + } + if (!ndef) + goto done; + + bw = (ndef + 31u) / 32u; + need = (uint64_t)n * nv * 4u * bw; + if (need > SPLIT_MAX_WORDS) + goto done; + + in = calloc((size_t)need, sizeof *in); + tmp = calloc((size_t)nv * 4u * bw, sizeof *tmp); + parent = malloc(ndef * sizeof *parent); + if (!in || !tmp || !parent) + goto done; + for (i = 0; i < ndef; i++) + parent[i] = (uint32_t)i; + + /* Reaching definitions, per value and per component. */ + do { + changed = 0; + if (++iters > SPLIT_MAX_ITERS) + goto done; + for (i = 0; i < n; i++) { + const struct uir_insn *ni = &s->insns[i]; + size_t sv[2], w, base = i * nv * 4u * bw; + unsigned ns; + + memcpy(tmp, in + base, + (size_t)nv * 4u * bw * sizeof *tmp); + if (defid[i] != NOREG) { + uint32_t d = ni->dst.index; + + for (k = 0; k < 4; k++) { + uint32_t *q; + + if (!((ni->writemask >> k) & 1)) + continue; + q = tmp + ((size_t)d * 4u + k) * bw; + memset(q, 0, bw * sizeof *q); + q[defid[i] / 32u] |= + 1u << (defid[i] % 32u); + } + } + ns = insn_succs(s, lbl, i, sv); + for (j = 0; j < ns; j++) { + uint32_t *dst = in + sv[j] * nv * 4u * bw; + + for (w = 0; w < (size_t)nv * 4u * bw; w++) + if ((dst[w] | tmp[w]) != dst[w]) { + dst[w] |= tmp[w]; + changed = 1; + } + } + } + } while (changed); + + /* Every definition a use can read becomes one value. */ + for (i = 0; i < n; i++) { + const struct uir_insn *ni = &s->insns[i]; + + for (j = 0; j < ni->nsrc; j++) { + unsigned rm = src_rmask(ni, j); + uint32_t v = ni->src[j].index, first = NOREG; + + if (ni->src[j].cls != UIR_REG_VIRT || v >= nv) + continue; + for (k = 0; k < 4; k++) { + const uint32_t *q; + uint32_t b; + + if (!((rm >> k) & 1)) + continue; + q = in + (i * nv * 4u + + (size_t)v * 4u + chan(ni->src[j].swizzle, k)) * bw; + for (b = 0; b < ndef; b++) { + if (!((q[b / 32u] >> (b % 32u)) & 1)) + continue; + if (first == NOREG) + first = b; + else + uf_union(parent, first, b); + } + } + } + } + + /* A value some definition writes only part of is left whole. Splitting + * such a value hands its webs registers of their own, but a partial + * write leaves the components it did not touch belonging to whichever + * web wrote them, and the interval model does not describe that - the + * webs overlap in a way the allocator cannot see, and glamor's + * composite rendered its early-return arm unconditionally. Bisected to + * one value (SGX_SPLIT_MAX), and every definition of it was a single + * component: `rcp.f32.x v5` then `rcp.f32.y v5`, read as `.xy`. + * + * Values written whole - which is what the front end's reused + * temporaries are - still split, and that is where the registers come + * from. */ + if (!getenv("SGX_SPLIT_PARTIAL")) { + uint32_t a, b; + + for (a = 0; a < nv; a++) { + uint32_t firstd = NOREG; + int partial = 0; + + for (b = 0; b < n; b++) + if (defid[b] != NOREG && + s->insns[b].dst.index == a && + s->insns[b].writemask != 0xf) + partial = 1; + if (!partial) + continue; + for (b = 0; b < n; b++) + if (defid[b] != NOREG && + s->insns[b].dst.index == a) { + if (firstd == NOREG) + firstd = defid[b]; + else + uf_union(parent, firstd, + defid[b]); + } + } + } + /* SGX_SPLIT_MAX=n splits only the first n values the front end + * numbered and leaves the rest fused, which bisects a miscompile down + * to the one value whose splitting causes it. */ + { + const char *e = getenv("SGX_SPLIT_MAX"); + uint32_t lim = (e && *e) ? (uint32_t)strtoul(e, NULL, 0) : nv; + uint32_t a, b; + + for (a = lim; a < nv; a++) { + uint32_t firstd = NOREG; + + for (b = 0; b < n; b++) + if (defid[b] != NOREG && + s->insns[b].dst.index == a) { + if (firstd == NOREG) + firstd = defid[b]; + else + uf_union(parent, firstd, + defid[b]); + } + } + } + newidx = malloc(ndef * sizeof *newidx); + ins = malloc(n * sizeof *ins); + out = malloc(sizeof *out); + if (!newidx || !ins || !out) { + free(ins); free(out); ins = NULL; out = NULL; + goto done; + } + for (i = 0; i < ndef; i++) + newidx[i] = NOREG; + for (i = 0; i < ndef; i++) { + uint32_t r = uf_find(parent, (uint32_t)i); + + if (newidx[r] == NOREG) + newidx[r] = nnew++; + } + + memcpy(ins, s->insns, n * sizeof *ins); + for (i = 0; i < n; i++) { + struct uir_insn *ni = &ins[i]; + + for (j = 0; j < ni->nsrc; j++) { + unsigned rm = src_rmask(&s->insns[i], j); + uint32_t v = ni->src[j].index, first = NOREG; + + if (ni->src[j].cls != UIR_REG_VIRT || v >= nv) + continue; + for (k = 0; k < 4 && first == NOREG; k++) { + const uint32_t *q; + uint32_t b; + + if (!((rm >> k) & 1)) + continue; + q = in + (i * nv * 4u + + (size_t)v * 4u + chan(ni->src[j].swizzle, k)) * bw; + for (b = 0; b < ndef; b++) + if ((q[b / 32u] >> (b % 32u)) & 1) { + first = b; + break; + } + } + /* Read before anything wrote it: keep a value of its + * own so the backend still names it in the diagnostic + * rather than silently aliasing another. */ + ni->src[j].index = first == NOREG ? nnew++ : + newidx[uf_find(parent, first)]; + } + if (defid[i] != NOREG) + ni->dst.index = newidx[uf_find(parent, defid[i])]; + } + + *out = *s; + out->insns = ins; + out->insns_cap = n; + out->nvirt = nnew; + ins = NULL; +done: + free(defid); + free(lbl); + free(in); + free(tmp); + free(parent); + free(newidx); + free(ins); + return out; +} + +/* scan() allocates the per-value arrays, so trying a second shader form means + * releasing the first. */ +static void free_scan(struct cg *c) +{ + free(c->vbase); + free(c->vfirst); + free(c->vlast); + free(c->vwidth); + free(c->vused); + c->vbase = NULL; + c->vfirst = NULL; + c->vlast = NULL; + c->vwidth = NULL; + c->vused = NULL; +} + +/* Uniform indices arrive one at a time during the scan, so the two per-uniform + * arrays grow to the highest index seen rather than being sized up front. */ +static int uni_room(struct cg *c, unsigned index) +{ + unsigned want = index + 1, cap = c->uni_cap; + uint8_t *p; + + if (want <= c->uni_cap) + return 1; + while (cap < want) + cap = cap ? 2u * cap : 8u; + p = realloc(c->uni_reach, cap); + if (!p) { + fail(c, "out of memory"); + return 0; + } + memset(p + c->uni_cap, 0, cap - c->uni_cap); + c->uni_reach = p; + p = realloc(c->uni_umask, cap); + if (!p) { + fail(c, "out of memory"); + return 0; + } + memset(p + c->uni_cap, 0, cap - c->uni_cap); + c->uni_umask = p; + p = realloc(c->uni_base, cap); + if (!p) { + fail(c, "out of memory"); + return 0; + } + memset(p + c->uni_cap, 0, cap - c->uni_cap); + c->uni_base = p; + c->uni_cap = cap; + return 1; +} + +static void scan(struct cg *c) +{ + const struct uir_shader *s = c->s; + size_t i, j; + unsigned k; + + c->nvirt = s->nvirt; + c->vbase = malloc((c->nvirt + 1) * sizeof *c->vbase); + c->vfirst = malloc((c->nvirt + 1) * sizeof *c->vfirst); + c->vlast = malloc((c->nvirt + 1) * sizeof *c->vlast); + c->vwidth = malloc((c->nvirt + 1) * sizeof *c->vwidth); + c->vused = calloc(c->nvirt + 1, sizeof *c->vused); + if (!c->vbase || !c->vfirst || !c->vlast || !c->vwidth || !c->vused) { + fail(c, "out of memory"); + return; + } + for (i = 0; i < c->nvirt; i++) { + c->vbase[i] = NOREG; + c->vfirst[i] = NOREG; + c->vlast[i] = 0; + c->vwidth[i] = 1; + } + + for (i = 0; i < s->ninsns; i++) { + const struct uir_insn *n = &s->insns[i]; + + if (n->dst.cls == UIR_REG_VIRT && n->dst.index < c->nvirt) { + unsigned w = n->writemask & 8 ? 4 : (n->writemask & 4 ? 3 : + (n->writemask & 2 ? 2 : 1)); + + if (c->vfirst[n->dst.index] == NOREG) + c->vfirst[n->dst.index] = (uint32_t)i; + c->vlast[n->dst.index] = (uint32_t)i; + if (w > c->vwidth[n->dst.index]) + c->vwidth[n->dst.index] = (uint8_t)w; + c->vused[n->dst.index] |= (uint8_t)n->writemask; + } + if (n->dst.cls == UIR_REG_OUT && n->dst.index + 1 > c->nout) + c->nout = n->dst.index + 1; + for (j = 0; j < n->nsrc; j++) { + const struct uir_ref *r = &n->src[j]; + + switch (r->cls) { + case UIR_REG_VIRT: + if (r->index < c->nvirt) { + /* A two-dimensional sample takes x + * and y of its coordinate; only a + * projected one goes on to read w. */ + unsigned rm = + elementwise(n->op) ? + n->writemask : + (n->op == UIR_TEX && j == 0) ? + 0x3 : 0xf; + unsigned w = swz_reach(r, rm); + + if (c->vfirst[r->index] == NOREG) + c->vfirst[r->index] = (uint32_t)i; + c->vlast[r->index] = (uint32_t)i; + if (w > c->vwidth[r->index]) + c->vwidth[r->index] = (uint8_t)w; + for (k = 0; k < 4; k++) + if ((rm >> k) & 1) + c->vused[r->index] |= + (uint8_t)(1u << chan(r->swizzle, k)); + } + break; + case UIR_REG_IN: + if (r->index + 1 > c->nin) c->nin = r->index + 1; + break; + case UIR_REG_TEXCOORD: + if (r->index + 1 > c->ntc) c->ntc = r->index + 1; + break; + case UIR_REG_UNIFORM: { + unsigned rm = elementwise(n->op) ? + n->writemask : 0xfu; + unsigned w = swz_reach(r, rm ? rm : 0xfu); + + if (r->index + 1 > c->nuni) + c->nuni = r->index + 1; + if (!uni_room(c, r->index)) + return; + if (w > c->uni_reach[r->index]) + c->uni_reach[r->index] = (uint8_t)w; + c->uni_umask[r->index] |= + (uint8_t)swz_mask(r, rm ? rm : 0xfu); + break; + } + case UIR_REG_SAMPLER: + if (r->index + 1 > c->nsamp) c->nsamp = r->index + 1; + break; + /* Built here rather than lazily in resolve() so + * alloc_regs() knows what the pool costs before it + * places it, and can spill what will not fit. */ + case UIR_REG_IMM: { + unsigned hw, hwswz, comp; + + /* Only where resolve() will read it from the + * bank: a vertex program materialises its + * literals instead, and giving it a pool here + * made the frame refuse the upload. */ + if (c->vtx_imm || + (!c->fragment && c->vtx_no_pool)) + break; + if (!imm_hw_const(r, c->fragment, &hw, &hwswz)) + c->pool_reads[pool_add_c(c, r->imm, + &comp)]++; + break; + } + default: + break; + } + } + } + + for (i = 0; i < s->nbindings; i++) { + const struct uir_binding *b = &s->bindings[i]; + + switch (b->cls) { + case UIR_REG_IN: if (b->index + 1 > c->nin) c->nin = b->index + 1; break; + case UIR_REG_TEXCOORD: if (b->index + 1 > c->ntc) c->ntc = b->index + 1; break; + case UIR_REG_UNIFORM: if (b->index + 1 > c->nuni) c->nuni = b->index + 1; break; + case UIR_REG_OUT: if (b->index + 1 > c->nout) c->nout = b->index + 1; break; + default: break; + } + } + + /* A forward branch stays inside the linear range it spans, but a + * backward one means a value defined late can be read again early, so + * anything live inside a loop is extended to cover the whole loop. */ + for (i = 0; i < s->ninsns; i++) { + const struct uir_insn *n = &s->insns[i]; + size_t lo; + int found = 0; + + if (n->op != UIR_BR && n->op != UIR_BRC) + continue; + for (lo = 0; lo < i; lo++) + if (s->insns[lo].op == UIR_LABEL && + s->insns[lo].label_id == n->branch_target) { + found = 1; + break; + } + if (!found) + continue; + for (j = 0; j < c->nvirt; j++) { + if (c->vfirst[j] == NOREG) + continue; + if (c->vlast[j] < lo || c->vfirst[j] > i) + continue; + if (c->vfirst[j] > lo) c->vfirst[j] = (uint32_t)lo; + if (c->vlast[j] < i) c->vlast[j] = (uint32_t)i; + } + } +} + +/* How many materialising slots the program actually reaches. The area above + * the values was sized for the most an instruction could possibly ask for - + * four quads, sixteen registers - and a program whose busiest instruction + * materialises one paid for four all the same. The task's temporary count is + * five bits, so sixteen registers of headroom is the difference between a + * program the part can run and one it refuses. + * + * Counted the way materialise_uniforms() spends them: one slot per distinct + * uniform an instruction reads, one per immediate source. */ +static unsigned max_materialised(const struct cg *c) +{ + unsigned i, best = 0; + + if ((!c->vtx_imm && !c->imm_temp) || !c->s) + return 0; + for (i = 0; i < c->s->ninsns; i++) { + const struct uir_insn *n = &c->s->insns[i]; + unsigned seen[UIR_MAX_SRC], nseen = 0, used = 0, sx, k; + + for (sx = 0; sx < n->nsrc && sx < UIR_MAX_SRC; sx++) { + if (n->src[sx].cls == UIR_REG_IMM) { + used++; + continue; + } + if (!c->vtx_limm || + n->src[sx].cls != UIR_REG_UNIFORM) + continue; + for (k = 0; k < nseen; k++) + if (seen[k] == n->src[sx].index) + break; + if (k < nseen) + continue; /* the same one twice */ + if (nseen < UIR_MAX_SRC) + seen[nseen++] = n->src[sx].index; + used++; + } + if (used > best) + best = used; + } + return best > UIR_MAX_SRC ? UIR_MAX_SRC : best; +} + +/* Scratch is claimed and released within one instruction, so it does not need + * registers of its own stacked above every value: at each instruction it can + * sit in whatever the values live there are not using. Returns 0 if there is + * no room at some instruction, which is a real refusal. */ +static int place_scratch(struct cg *c, uint32_t *peak, uint32_t budget) +{ + size_t n = c->s ? c->s->ninsns : 0, i; + uint32_t words = (budget + 31u) / 32u, top = *peak, smax = top; + uint32_t v, t, q, *occ; + + free(c->scratch_at); + c->scratch_at = NULL; + c->scratch_placed = 0; + c->outreg_hole = NOREG; + c->fixed_top = top; + c->scratch = top; + /* Off by default. Placing scratch in the holes between live values is + * sound in principle and wrong somewhere in practice: glamor's masked + * composite - two samplers, each correct on its own - comes back white + * with it and correct without, and so does the feature suite's + * glamormask and glamor composite, 85 of 99 against 83. The live + * intervals are not the reason; SGX_CHECK_LIVE reports every one of + * them covering its range. So something occupies a register that this + * occupancy map does not model, and until that is found the + * optimisation is not worth what it breaks: measured, it buys nothing + * anyway - glmark2-drm build 14 and 14, shading 15 and 15, texture 31 + * against 29, ioquake3 9.6 either way. SGX_SCRATCH_PLACE turns it back + * on. */ + if (!c->scratch_regs || !n || !getenv("SGX_SCRATCH_PLACE")) { + *peak = top + c->scratch_regs; + return 1; + } + occ = calloc(n * words, sizeof *occ); + c->scratch_at = malloc(n * SCRATCH_QUADS * sizeof *c->scratch_at); + if (!c->scratch_need) + c->scratch_need = calloc(n, sizeof *c->scratch_need); + /* Nothing measured on the first pass, so the quads keep the packed + * block and this does nothing until there is something to place. */ + if (!c->scratch_w) { + c->scratch_w = calloc(n * SCRATCH_QUADS, sizeof *c->scratch_w); + free(occ); + free(c->scratch_at); + c->scratch_at = NULL; + *peak = top + c->scratch_regs; + return 1; + } + if (!occ || !c->scratch_at) { + free(occ); + free(c->scratch_at); + c->scratch_at = NULL; + *peak = top + c->scratch_regs; + return 1; /* placing it is an optimisation, not a duty */ + } + for (v = 0; v < c->nvirt; v++) { + if (c->vfirst[v] == NOREG || c->vbase[v] == NOREG) + continue; + for (t = c->vfirst[v]; t <= c->vlast[v] && t < n; t++) + for (q = 0; q < c->vwidth[v]; q++) { + uint32_t r = c->vbase[v] + q; + + if (r < budget) + occ[t * words + r / 32u] |= + 1u << (r % 32u); + } + } + /* The materialising slots live for the whole program. An unpacked + * input does not: it is unpacked at entry and read until its last use, + * and after that its quad is as free as any dead value's. Holding it + * to the end kept the output stand-in from ever finding a hole. */ + { + uint32_t last_in[16], b; + + for (b = 0; b < 16; b++) + last_in[b] = 0; + for (i = 0; i < n; i++) { + const struct uir_insn *ni = &c->s->insns[i]; + unsigned j; + + for (j = 0; j < ni->nsrc; j++) + if (ni->src[j].cls == UIR_REG_IN && + ni->src[j].index < 16) + last_in[ni->src[j].index] = + (uint32_t)i; + } + for (i = 0; i < n; i++) + for (q = c->reg_values; q < top && q < budget; q++) { + uint32_t live = 1, b2; + + for (b2 = 0; b2 < 16; b2++) + if (c->in_unpack[b2] != NOREG && + q >= c->in_unpack[b2] && + q < c->in_unpack[b2] + QUAD) { + live = (uint32_t)i <= + last_in[b2]; + break; + } + if (live) + occ[i * words + q / 32u] |= + 1u << (q % 32u); + } + } + + /* The fragment output stand-in and the discard condition are live over + * a known stretch too - the output from the first write to the end, + * the discard from its first kill - so they go in holes as well rather + * than on top of everything. */ + if (c->fragment) { + uint32_t first_out = (uint32_t)n, i2, base, k; + + for (i2 = 0; i2 < n; i2++) + if (c->s->insns[i2].dst.cls == UIR_REG_OUT || + c->s->insns[i2].op == UIR_EMIT) { + first_out = i2; + break; + } + for (base = 0; base + QUAD <= budget; base++) { + uint32_t ok = 1; + + for (i2 = first_out; i2 < n && ok; i2++) + for (k = 0; k < QUAD; k++) + if (occ[i2 * words + + (base + k) / 32u] & + (1u << ((base + k) % 32u))) { + ok = 0; + break; + } + if (ok) + break; + } + if (base + QUAD <= budget) { + c->outreg_hole = base; + for (i2 = first_out; i2 < n; i2++) + for (k = 0; k < QUAD; k++) + occ[i2 * words + (base + k) / 32u] |= + 1u << ((base + k) % 32u); + if (base + QUAD > smax) + smax = base + QUAD; + } + } + for (i = 0; i < n; i++) { + unsigned q; + + for (q = 0; q < SCRATCH_QUADS; q++) { + uint32_t base, need = c->scratch_w ? + c->scratch_w[i * SCRATCH_QUADS + q] : 0; + uint32_t k; + + c->scratch_at[i * SCRATCH_QUADS + q] = 0; + if (!need) + continue; + for (base = 0; base + need <= budget; base++) { + uint32_t ok = 1; + + for (k = 0; k < need; k++) + if (occ[i * words + + (base + k) / 32u] & + (1u << ((base + k) % 32u))) { + ok = 0; + break; + } + if (ok) + break; + } + if (base + need > budget) { + free(occ); + return 0; + } + c->scratch_at[i * SCRATCH_QUADS + q] = base; + /* Taken, so the next quad of this instruction goes + * somewhere else. */ + for (k = 0; k < need; k++) + occ[i * words + (base + k) / 32u] |= + 1u << ((base + k) % 32u); + if (base + need > smax) + smax = base + need; + } + } + c->scratch_placed = 1; + free(occ); + c->scratch = c->scratch_at[0]; + *peak = smax > top ? smax : top; + return 1; +} + +/* Does each value's linear interval actually cover where it is live? The + * allocator reuses a register the moment one interval ends, so an interval + * that is not a superset of the real live range hands two live values the same + * register. On straight-line code [first, last] is trivially a superset; with + * branches it need not be, and that is the shape that miscompiles. + * SGX_CHECK_LIVE reports every value it is wrong for. */ +static void check_live(struct cg *c) +{ + const struct uir_shader *s = c->s; + size_t n = s->ninsns, i; + uint32_t *lbl, *live, nv = c->nvirt, w, bw; + int changed, bad = 0; + + if (!nv || !n) + return; + bw = (nv + 31u) / 32u; + lbl = malloc(((size_t)s->next_label + 1) * sizeof *lbl); + live = calloc((n + 1) * bw, sizeof *live); /* LIVE_IN per insn */ + if (!lbl || !live) { + free(lbl); + free(live); + return; + } + for (i = 0; i <= s->next_label; i++) + lbl[i] = NOREG; + for (i = 0; i < n; i++) + if (s->insns[i].op == UIR_LABEL && + s->insns[i].label_id <= s->next_label) + lbl[s->insns[i].label_id] = (uint32_t)i; + + do { + changed = 0; + for (i = n; i-- > 0; ) { + const struct uir_insn *ni = &s->insns[i]; + uint32_t *in = live + i * bw, out[64], j; + size_t sv[2]; + unsigned ns, k; + + if (bw > 64) + goto done; + memset(out, 0, bw * sizeof *out); + ns = insn_succs(s, lbl, i, sv); + for (k = 0; k < ns; k++) + for (j = 0; j < bw; j++) + out[j] |= live[sv[k] * bw + j]; + if (ni->dst.cls == UIR_REG_VIRT && ni->dst.index < nv && + ni->writemask == 0xf) + out[ni->dst.index / 32u] &= + ~(1u << (ni->dst.index % 32u)); + for (k = 0; k < ni->nsrc; k++) + if (ni->src[k].cls == UIR_REG_VIRT && + ni->src[k].index < nv) + out[ni->src[k].index / 32u] |= + 1u << (ni->src[k].index % 32u); + for (j = 0; j < bw; j++) + if ((in[j] | out[j]) != in[j]) { + in[j] |= out[j]; + changed = 1; + } + } + } while (changed); + + for (w = 0; w < nv; w++) { + if (c->vfirst[w] == NOREG) + continue; + for (i = 0; i < n; i++) + /* Only past the end matters: the allocator hands the + * register to the next value the moment vlast passes, + * so a value still live after it is the hazard. Live + * before vfirst is read-modify-write on components + * nothing has written yet, which is harmless. */ + if ((live[i * bw + w / 32u] >> (w % 32u)) & 1 && + (uint32_t)i > c->vlast[w]) { + fprintf(stderr, "sgx: check_live: v%u live at " + "%u but its interval is [%u..%u]\n", + w, (unsigned)i, c->vfirst[w], + c->vlast[w]); + bad++; + break; + } + } + if (!bad) + fprintf(stderr, "sgx: check_live: %u value(s), every interval " + "covers its live range\n", nv); +done: + free(lbl); + free(live); +} + +/* The register budget a pass is held to. + * + * The measuring pass is meant to be the relaxed one - how much scratch the + * code reaches is only known once it has been emitted - but it was handed + * DEFAULT_TEMPS outright, which is stricter than the caller's own count + * whenever that is larger. A program needing sixty-eight of the ninety-six + * the driver allows was refused by the loose pass and never reached the + * strict one: glmark2's terrain noise shader, to the register. */ +static uint32_t cg_budget(const struct cg *c) +{ + uint32_t want = c->opts.max_temps ? (uint32_t)c->opts.max_temps : + DEFAULT_TEMPS; + + if (c->relax_budget && want < DEFAULT_TEMPS) + want = DEFAULT_TEMPS; + return want; +} + +/* Everything placed above the values - the uniform staging quads, the unpacked + * inputs, the output stand-in, the discard flag - has to fit the same budget. + * Growing past it unchecked produced a register number the encoding cannot + * hold, reported half a shader later as "past the 127 the encoding holds" + * rather than as the allocation failure it is. */ +static int take_regs(struct cg *c, uint32_t *peak, uint32_t n, uint32_t budget, + const char *what) +{ + if (*peak + n > budget) { + fail(c, "out of temporary registers: the %s needs %u more " + "than the %u available", what, n, budget); + return 0; + } + *peak += n; + return 1; +} + +static int alloc_regs(struct cg *c) +{ + /* The measuring pass is not held to the hardware's count: how much + * scratch the code reaches is only known once it has been emitted, and + * refusing a shader against the worst-case estimate turned away + * programs that fit comfortably once the real figure was in. */ + uint32_t budget = cg_budget(c); + uint32_t *freeat; + uint32_t i, k, ix, next = 0, peak = 0; + + /* The sa bank holds 128 registers and a large program can want more: + * shed the least-read literals into the instruction stream until what + * is left fits. Decided here, ahead of the staging quads a + * materialised literal needs. */ + { + unsigned base = uni_span(c) + SMP_SLOT * smp_slots_total(c); + uint32_t kept = c->npool; + + memset(c->pool_spill, 0, sizeof c->pool_spill); + while (kept && base + QUAD * kept > UIR_SA_BANK) { + uint32_t worst = c->npool; + + for (i = 0; i < c->npool; i++) + if (!c->pool_spill[i] && + (worst == c->npool || + c->pool_reads[i] < c->pool_reads[worst])) + worst = i; + if (worst == c->npool) + break; + c->pool_spill[worst] = 1; + kept--; + } + c->imm_temp = kept != c->npool; + } + + /* Texture coordinates follow the inputs; with explicit bases the + * inputs are not four registers apart, so it is the end of the last + * one that they follow. */ + c->tc_base = QUAD * c->nin; + if (c->frag_in_bases && c->nin) { + /* frag_in_base holds one entry per input the frame can carry; + * nin is whatever index the shader read highest, so a program + * naming IN[16] and up indexed it off the end. */ + if (c->nin > sizeof c->opts.frag_in_base / + sizeof c->opts.frag_in_base[0]) { + fail(c, "%u fragment inputs, the frame carries %u", + c->nin, (unsigned)(sizeof c->opts.frag_in_base / + sizeof c->opts.frag_in_base[0])); + return -1; + } + c->tc_base = (uint32_t)c->opts.frag_in_base[c->nin - 1] + QUAD; + } + + freeat = calloc(budget + 1, sizeof *freeat); + if (!freeat) { fail(c, "out of memory"); return -1; } + for (i = 0; i < budget; i++) + freeat[i] = NOREG; + + /* Values take only as many registers as any use of them reaches. */ + { + static int noreuse = -1; + + if (noreuse < 0) + noreuse = getenv("SGX_NO_REUSE") != NULL; + /* SGX_NO_REUSE gives every value a register of its own. It + * tells a wrong live range from a wrong renumbering: with no + * reuse an interval cannot be too short to matter. */ + if (noreuse) + for (i = 0; i < c->nvirt; i++) + if (c->vfirst[i] != NOREG) { + if (next + c->vwidth[i] > budget) + break; + c->vbase[i] = next; + next += c->vwidth[i]; + if (next > peak) + peak = next; + } + } + /* In order of where each value starts, not by its number. The scan + * hands a register on when the last value in it died, and `freeat` + * remembers only that one - so a value considered out of order can be + * placed in a register whose earlier occupant is still live. Splitting + * makes that reachable: it turns six values into ninety-one, and their + * numbers no longer run in the order their ranges begin. */ + if (!c->vorder) + c->vorder = malloc((c->nvirt + 1) * sizeof *c->vorder); + if (c->vorder) { + uint32_t a, b, t; + + for (a = 0; a < c->nvirt; a++) + c->vorder[a] = a; + for (a = 1; a < c->nvirt; a++) /* insertion sort */ + for (b = a; b && c->vfirst[c->vorder[b]] < + c->vfirst[c->vorder[b - 1]]; b--) { + t = c->vorder[b]; + c->vorder[b] = c->vorder[b - 1]; + c->vorder[b - 1] = t; + } + } + for (ix = 0; ix < c->nvirt; ix++) { + uint32_t w; + + i = c->vorder ? c->vorder[ix] : ix; + w = c->vwidth[i]; + if (c->vfirst[i] == NOREG || c->vbase[i] != NOREG) + continue; + for (k = 0; k + w <= budget; k++) { + uint32_t q, ok = 1; + + for (q = 0; q < w; q++) + if (freeat[k + q] != NOREG && freeat[k + q] >= c->vfirst[i]) + ok = 0; + if (!ok) + continue; + c->vbase[i] = k; + for (q = 0; q < w; q++) + freeat[k + q] = c->vlast[i]; + if (k + w > peak) + peak = k + w; + break; + } + if (c->vbase[i] == NOREG) { + fail(c, "out of temporary registers: v%u needs %u of the " + "%u available (raise uir_codegen_opts.max_temps)", + i, w, budget); + free(freeat); + return -1; + } + } + free(freeat); + + /* Scratch and the fragment output stand-in sit above the allocated + * values, so a shader that needs neither pays for neither. */ + { + uint32_t vpeak = peak; /* what the values themselves took */ + + c->reg_values = vpeak; + } + /* Three scratch quads, not one per uniform: an instruction reads at + * most three of them, so this is what a program of any uniform count + * costs. Holding them all at once needed four temporaries each, and + * the task's temporary count is five bits. */ + if (c->vtx_imm || c->imm_temp) { + unsigned nm = getenv("SGX_IMM_WORST") ? UIR_MAX_SRC : + max_materialised(c); + + c->uni_temp = peak; + if (!take_regs(c, &peak, QUAD * nm, budget, + "uniform staging quads")) + return -1; + } + /* A quad per packed or half-float input, unpacked into at entry. */ + for (i = 0; i < 16; i++) + c->in_unpack[i] = NOREG; + if (c->fragment && (c->opts.frag_in_packed | c->opts.frag_in_f16)) + for (i = 0; i < c->nin && i < 16; i++) + if ((c->opts.frag_in_packed | c->opts.frag_in_f16) & + (1u << i)) { + c->in_unpack[i] = peak; + if (!take_regs(c, &peak, QUAD, budget, + "unpacked input")) + return -1; + } + if (!place_scratch(c, &peak, budget)) { + fail(c, "out of temporary registers: no room for the %u " + "scratch register(s) an instruction needs within %u", + c->scratch_regs, budget); + return -1; + } + if (c->fragment) { + if (c->outreg_hole != NOREG) { + c->fragout = c->outreg_hole; + } else { + c->fragout = peak; + if (!take_regs(c, &peak, QUAD, budget, + "fragment output stand-in")) + return -1; + } + /* A second declared output is a dual source blend's other + * colour. It needs a quad of its own, and it is not written + * to the pixel: the blend names this register. */ + if (c->nout > 1u) { + c->fragout1 = peak; + c->has_fragout1 = 1; + if (!take_regs(c, &peak, QUAD, budget, + "second fragment colour")) + return -1; + } + } + /* One register, not a quad: the discard writes its condition with a + * single-component mask and the test names the same register as a + * suppressed destination, so the other three were never touched. */ + if (c->uses_kill) { + c->killreg = peak; + if (!take_regs(c, &peak, 1u, budget, "discard flag")) + return -1; + } + if (getenv("SGX_REG_BREAKDOWN")) { + uint32_t v, reach = 0, live = 0, nv = 0; + + /* What the values cost against what they use: a value is given + * as many registers as its highest component reaches, so one + * touched only at .w costs four. The difference is what + * compacting the components would return. */ + for (v = 0; v < c->nvirt; v++) { + unsigned m = c->vused[v], b, pc = 0; + + if (c->vfirst[v] == NOREG) + continue; + nv++; + reach += c->vwidth[v]; + for (b = 0; b < 4; b++) + pc += (m >> b) & 1; + live += pc ? pc : 1; + } + fprintf(stderr, "sgx: regs: values %u, unpack+imm %u, " + "scratch %u, out %u, discard %u = %u (budget %u)\n", + c->reg_values, c->fixed_top - c->reg_values, + c->scratch_regs, c->fragment ? QUAD : 0u, + c->uses_kill ? 1u : 0u, peak, budget); + fprintf(stderr, "sgx: regs: %u value(s), reach %u register(s), " + "components actually used %u\n", nv, reach, live); + if (c->pairfix) + fprintf(stderr, "sgx: regs: %u padding instruction(s) " + "to split branch pairs\n", c->pairfix); + if (getenv("SGX_REG_VALUES")) + for (v = 0; v < c->nvirt; v++) { + if (c->vfirst[v] == NOREG) + continue; + fprintf(stderr, "sgx: regs: v%-3u width %u " + "used %x live [%u..%u] at %u\n", + v, c->vwidth[v], c->vused[v], + c->vfirst[v], c->vlast[v], + c->vbase[v]); + } + } + if (peak > budget) { + fail(c, "out of temporary registers: %u needed, %u available " + "(values reach %u, scratch %u, output %u, discard %u)", + peak, budget, c->fixed_top, c->scratch_regs, + c->fragment ? QUAD : 0u, + c->uses_kill ? 1u : 0u); + return -1; + } + + for (i = 0; i < c->nin; i++) + note_map(c, UIR_REG_IN, i, QUAD * i); + for (i = 0; i < c->ntc; i++) + note_map(c, UIR_REG_TEXCOORD, i, c->tc_base + QUAD * i); + /* Uniforms take the registers they are read through, not a quad each. + * Six of alacritty's seven are scalars, which is twenty-eight + * registers of a thirty-two register bank spent on eight registers of + * values. The driver gathers them into this layout, so the two must + * agree. The vertex path keeps the quad stride: its uniforms are + * primary attributes the vertex PDS brings in as one block. */ + { + unsigned b = 0; + + int nopack = getenv("SGX_NO_UNI_PACK") != NULL; + + if (c->nuni && !uni_room(c, c->nuni - 1)) + return -1; + for (i = 0; i < c->nuni; i++) { + unsigned w = (!nopack && c->uni_reach[i]) ? + c->uni_reach[i] : (unsigned)QUAD; + + if (!c->fragment && c->vtx_uniform_pa_base) { + note_map(c, UIR_REG_UNIFORM, i, + c->vtx_uniform_pa_base + QUAD * i); + continue; + } + c->uni_base[i] = (uint8_t)b; + note_map(c, UIR_REG_UNIFORM, i, b); + b += w; + } + c->uni_regs = b; + if (getenv("SGX_DUMP_UNI")) + for (i = 0; i < c->nuni; i++) + fprintf(stderr, "sgx: uniform %u reach %u at " + "%u\n", i, c->uni_reach[i], + c->uni_base[i]); + } + for (i = 0; i < c->nout; i++) + note_map(c, UIR_REG_OUT, i, + QUAD * ((c->vtx_out_slot && i < 16 && !c->fragment) ? + c->vtx_out_slot[i] : i)); + /* sa layout: uniforms, then one texture state block per sampler - three + * dwords used of the quad reserved, {ctl, fmt, addr}, consecutive + * because the sample names only the first - then + * the literal pool. Sampler blocks are sized at one quad; the real + * descriptor width is not established, see README. */ + { + /* Vertex uniforms that became primary attributes leave no + * hole in the sa bank, so what follows starts at zero. */ + unsigned usa = c->uni_regs; + + /* Three registers per sampler, not a quad: the block is + * {ctl, fmt, addr} and the sample names only the first, so the + * fourth was reserved and never read. The bank holds + * thirty-two registers and the X server's masked composite + * wanted thirty-six; two samplers give back exactly the four + * it was over by. The driver writes the descriptors at this + * same stride - see ctx_write_sampler_slots(). */ + c->smp_base = usa; + for (i = 0; i < c->nsamp; i++) + note_map(c, UIR_REG_SAMPLER, i, + usa + SMP_SLOT * smp_slot_of(c, i)); + c->pool_base = usa + SMP_SLOT * smp_slots_total(c); + } + if (c->lay) + c->lay->pool_base = c->pool_base; + /* Densely over the entries the bank keeps: a spilled one is written + * in front of the instruction that reads it and needs no register. */ + { + uint32_t b = c->pool_base; + + for (i = 0; i < c->npool; i++) + if (!c->pool_spill[i]) { + c->pool[i].sa_base = b; + b += QUAD; + } + c->nkept = (b - c->pool_base) / QUAD; + c->nsa = b; + } + return 0; +} + + +/* Unpack every packed input into its quad, before the program runs. + * + * The frame can hand a varying over as one packed 8888 dword in a primary + * attribute instead of iterating it as a coordinate set, and it is far cheaper + * that way - the same scene renders in 5 ms against 106. A program that only + * moves the value to the output can read the dword as it stands; one that + * computes with it needs the four bytes as floats, which is the same unpack + * the pre-iterated texel gets. */ +/* Which predicate carries "this fragment survives", p0..p3 as 1..4. */ +#define SGX_PRED_SURVIVE 2u + +/* "This fragment survives" is written by the kill test, which sits wherever + * the discard is - so a path that never reaches a discard never writes it, and + * the closing move into o0 is predicated on whatever the register held. It + * happened to read true often enough to look right; alacritty's glyph passes, + * which discard only on their background pass, then wrote nothing once that + * pass started discarding as it should. Zero plus zero is positive-or-zero, so + * the same test the kill uses sets it true at entry. */ +static void emit_kill_init(struct cg *c) +{ + struct usse_insn proto; + + if (!c->fragment || !c->uses_kill) + return; + memset(&proto, 0, sizeof proto); + proto.op = USSE_TEST; + proto.skipinv = 1; + proto.mask = 1; + proto.test_type = USSE_TEST_POZ; + proto.test_alu = USSE_TEST_FADD; + proto.test_pdst = SGX_PRED_SURVIVE - 1; + proto.dst.bank = USSE_TEMP; + proto.dst.num = (uint8_t)c->killreg; + proto.src[1].bank = USSE_CONST; + proto.src[1].num = USSE_C_ZERO; + proto.src[2].bank = USSE_CONST; + proto.src[2].num = USSE_C_ZERO; + emit(c, &proto); +} + +static void emit_in_unpack(struct cg *c) +{ + /* Measured: reading x from the low byte instead swaps red and blue on + * every packed varying - "glamor mask" then wants ff0000 and gets + * 0000ff. The iterator packs in the texel's order. */ + static const uint8_t byte[4] = { 2, 1, 0, 3 }; + /* SGX_PACKED_ORDER=rgba reads x from the low byte instead, which says + * which order the record's colour slot is actually packed in. */ + static const uint8_t rgba[4] = { 0, 1, 2, 3 }; + const uint8_t *ord = (c->opts.frag_packed_rgba || + getenv("SGX_PACKED_ORDER")) ? rgba : byte; + struct usse_insn proto; + unsigned i, k; + + if (!c->fragment || !(c->opts.frag_in_packed | c->opts.frag_in_f16)) + return; + for (i = 0; i < c->nin && i < 16; i++) { + uint8_t src; + + if (c->in_unpack[i] == NOREG) + continue; + src = (uint8_t)(c->frag_in_bases ? + c->opts.frag_in_base[i] : QUAD * i); + /* SGX_PACKED_BASE sweeps the register the packed varying is + * read from, to find where the iterator actually delivers it. */ + { + const char *e = getenv("SGX_PACKED_BASE"); + + if (e && *e) + src = (uint8_t)strtoul(e, NULL, 0); + } + /* Half floats: two components a register, and the component + * field of a pack is a BYTE offset into the source, so the + * halves are at 0 and 2 - the vendor's encoder refuses + * anything else (usp/hwinst.c, "Byte-offset must be 0 or 2") + * and its own vertex unpack computes the same pair, + * (j * 2) >> 2 for the register and (j * 2) % 4 for the byte + * (opengles2/use.c:1516-1517). No scale: the value is already + * a float, not a normalised integer. */ + if (c->opts.frag_in_f16 & (1u << i)) { + for (k = 0; k < QUAD; k++) { + memset(&proto, 0, sizeof proto); + proto.op = USSE_UNPCKF32F16; + proto.mask = 1; + proto.bytemask = 0xF; + proto.dst.bank = USSE_TEMP; + proto.dst.num = (uint8_t)(c->in_unpack[i] + k); + proto.src[1].bank = USSE_PA; + proto.src[1].num = (uint8_t)(src + k / 2u); + proto.src[1].comp = (uint8_t)(2u * (k % 2u)); + proto.src[2] = proto.src[1]; + emit(c, &proto); + } + continue; + } + for (k = 0; k < QUAD; k++) { + memset(&proto, 0, sizeof proto); + proto.op = USSE_UNPCKF32U8; + proto.mask = 1; + proto.scale = 1; + proto.bytemask = 0xF; /* an f32 is written whole */ + proto.dst.bank = USSE_TEMP; + proto.dst.num = (uint8_t)(c->in_unpack[i] + k); + proto.src[1].bank = USSE_PA; + proto.src[1].num = src; + proto.src[1].comp = ord[k]; + proto.src[2] = proto.src[1]; + emit(c, &proto); + } + } +} + +/* --- lowering ---------------------------------------------------------- */ + +static void lower_saturate(struct cg *c, const struct opnd *d, unsigned mask) +{ + struct usse_insn proto; + struct opnd s[3]; + + memset(&proto, 0, sizeof proto); + memset(s, 0, sizeof s); + s[1] = *d; + s[1].swz = UIR_SWZ_XYZW; + s[2] = konst(USSE_C_ZERO, 0); + emit_vec(c, USSE_MAX, d, mask, s, 0x6, &proto); + s[2] = konst(USSE_C_ONE, 0); + emit_vec(c, USSE_MIN, d, mask, s, 0x6, &proto); +} + +static void lower_dot(struct cg *c, const struct uir_insn *n, unsigned terms) +{ + struct usse_insn proto; + struct opnd s[3], t, d; + uint16_t want[4]; + unsigned i, single; + + memset(&proto, 0, sizeof proto); + memset(s, 0, sizeof s); + s[1] = resolve(c, &n->src[0]); + s[2] = resolve(c, &n->src[1]); + d = resolve(c, &n->dst); + + /* A single-channel result can be accumulated straight into the + * destination register, as the vendor does; otherwise it lands in a + * scratch and is replicated. */ + single = popcount4(n->writemask) == 1; + if (single) { + unsigned k = n->writemask & 1 ? 0 : (n->writemask & 2 ? 1 : + (n->writemask & 4 ? 2 : 3)); + + t = d; + t.base = (uint8_t)(d.base + k); + } else { + t = scratch_q(c, 0); + } + + /* The dot accumulates over `terms` iterations into one register, so the + * destination must not advance. Confirmed by vendor code (f06-dot). */ + want[0] = MOE_INC0; + for (i = 0; i < 2; i++) + want[i + 2] = s[i + 1].swz == UIR_SWZ_XYZW ? MOE_INC1 + : MOE_SWZ(s[i + 1].swz); + want[1] = MOE_INC1; + moe_need(c, want); + + proto.repeat = (uint8_t)terms; + proto.op = USSE_DP; + proto.dst.bank = t.bank; + proto.dst.num = t.base; + proto.src[1].bank = s[1].bank; proto.src[1].num = s[1].base; + proto.src[1].neg = s[1].neg; proto.src[1].abs = s[1].abs; + proto.src[2].bank = s[2].bank; proto.src[2].num = s[2].base; + proto.src[2].neg = s[2].neg; proto.src[2].abs = s[2].abs; + emit(c, &proto); + if (single) + return; + + /* Replicate the scalar across the destination mask. */ + memset(&proto, 0, sizeof proto); + memset(s, 0, sizeof s); + s[1] = t; + s[1].swz = UIR_SWZ_XXXX; + emit_vec(c, USSE_MOV, &d, n->writemask, s, 0x2, &proto); +} + +/* dst = test(a - b) ? 1 : 0, or with invert the other way round. SGE is the + * inverted form: a >= b is !(a - b < 0). It used to be lowered as b - a < 0, + * which is a > b - the equal case came out 0, so step(x, x) and any + * comparison against the value's own bound went the wrong way. movc's .tn is + * a strict less-than-zero test (usedisasm.c maps EXTSIGNED to + * USEASM_TEST_LTZERO), and a - a is +0.0, which it does not take. */ +static void lower_cmpset(struct cg *c, const struct uir_insn *n, unsigned test, + int invert) +{ + struct usse_insn proto; + struct opnd s[3], t, d; + struct opnd a = resolve(c, &n->src[0]); + struct opnd b = resolve(c, &n->src[1]); + + memset(&proto, 0, sizeof proto); + memset(s, 0, sizeof s); + t = scratch_q(c, 0); + d = resolve(c, &n->dst); + + /* t = a - b */ + s[0] = legal_src0(c, a, 1, n->writemask); + s[1] = konst(USSE_C_ONE, 0); + s[2] = b; + s[2].neg = !s[2].neg; + emit_vec(c, USSE_MAD, &t, n->writemask, s, 0x7, &proto); + + memset(&proto, 0, sizeof proto); + memset(s, 0, sizeof s); + proto.movc_test = (uint8_t)test; + s[0] = t; + s[1] = konst(invert ? USSE_C_ZERO : USSE_C_ONE, 0); + s[2] = konst(invert ? USSE_C_ONE : USSE_C_ZERO, 0); + emit_vec(c, USSE_MOVC, &d, n->writemask, s, 0x7, &proto); +} + +/* Put the uniforms one instruction reads into scratch quads with LIMM, right + * in front of it. The values are left at zero: the driver patches them in as + * they change, which is why the map of instruction to uniform dword is + * reported rather than assumed. */ +static void materialise_uniforms(struct cg *c, const struct uir_insn *n) +{ + struct usse_insn in; + unsigned s, k; + + if (!c->vtx_imm && !c->imm_temp) + return; + for (s = 0; s < UIR_MAX_UNIFORM; s++) + c->uni_live[s] = NOREG; + for (s = 0; s < UIR_MAX_SRC; s++) + c->imm_live[s] = NOREG; + c->uni_used = 0; + c->cur_src = n->src; + + /* Uniforms move to temps only where they have no bank to live in, + * which is the vertex path: a fragment program reads them from the sa + * bank and materialises its spilled literals alone. */ + for (s = 0; c->vtx_limm && c->vtx_imm && s < n->nsrc; s++) { + unsigned u = n->src[s].index; + + if (n->src[s].cls != UIR_REG_UNIFORM) + continue; + if (u >= UIR_MAX_UNIFORM) { + fail(c, "uniform %u is past the %u this handles", u, + UIR_MAX_UNIFORM); + return; + } + if (c->uni_live[u] != NOREG) + continue; /* already in a scratch quad */ + if (c->uni_used >= UIR_MAX_SRC) { + fail(c, "more than %u distinct uniforms in one " + "instruction", UIR_MAX_SRC); + return; + } + c->uni_live[u] = c->uni_temp + QUAD * c->uni_used++; + for (k = 0; k < QUAD; k++) { + if (c->nlimm >= UIR_MAX_LIMM) { + fail(c, "more than %u materialised uniform " + "dwords", UIR_MAX_LIMM); + return; + } + memset(&in, 0, sizeof in); + in.op = USSE_LIMM; + in.dst.bank = USSE_TEMP; + in.dst.num = (uint8_t)(c->uni_live[u] + k); + in.imm = 0; + c->limm[c->nlimm].insn = (uint32_t)c->n; + c->limm[c->nlimm].dword = QUAD * u + k; + c->nlimm++; + emit(c, &in); + } + } + + /* An immediate goes the same way, and needs no patching later: its + * value is known here. This is what lets a vertex program use a + * constant other than the hardware's own 0.0 and 1.0. */ + for (s = 0; s < n->nsrc; s++) { + unsigned hw, hwswz, comp; + + if (n->src[s].cls != UIR_REG_IMM) + continue; + /* A vertex program has no pool and materialises every + * literal; a fragment one materialises only what the bank + * could not hold. */ + if (!c->vtx_imm) { + if (imm_hw_const(&n->src[s], c->fragment, &hw, &hwswz)) + continue; + if (!c->pool_spill[pool_add_c(c, n->src[s].imm, + &comp)]) + continue; + } + /* imm_live is indexed by the source number, so that is what + * bounds it - testing the count of materialised operands let + * a fifth source write past the array and over cur_src. */ + if (s >= UIR_MAX_SRC || c->uni_used >= UIR_MAX_SRC) { + fail(c, "more than %u materialised operands in one " + "instruction", UIR_MAX_SRC); + return; + } + c->imm_live[s] = c->uni_temp + QUAD * c->uni_used++; + for (k = 0; k < QUAD; k++) { + memset(&in, 0, sizeof in); + in.op = USSE_LIMM; + in.dst.bank = USSE_TEMP; + in.dst.num = (uint8_t)(c->imm_live[s] + k); + memcpy(&in.imm, &n->src[s].imm[k], 4); + emit(c, &in); + } + } +} + +/* Ends a fragment program: the packed pixel moves into o0. A shader that + * discards selects between it and o0's incoming value first - the destination, + * which the read-modify-write object type preloads - so a discarded fragment + * leaves the pixel it found. */ +static void emit_frag_tail(struct cg *c) +{ + struct usse_insn proto; + + /* Without .end the USSE runs on into whatever follows it in memory. */ + memset(&proto, 0, sizeof proto); + proto.op = USSE_MOV; + proto.mask = 1; + proto.dst.bank = USSE_OUT; + proto.dst.num = 0; + proto.src[1].bank = USSE_TEMP; + proto.src[1].num = (uint8_t)c->fragout; + proto.skipinv = c->uses_kill; + proto.pred = c->uses_kill ? SGX_PRED_SURVIVE : 0; + proto.end = !c->uses_kill; + emit(c, &proto); + if (!c->uses_kill) + return; + /* The program ends on its own instruction rather than on the + * predicated one: whether .end is honoured when the predicate is false + * is not established, and a program that runs off its end stalls the + * raster. */ + memset(&proto, 0, sizeof proto); + proto.op = USSE_NOP; + proto.end = 1; + emit(c, &proto); +} + +/* Record that the branch about to be emitted targets a label, so the offset + * can be filled in once every label's address is known. */ +static int branch_fixup(struct cg *c, uint32_t target, uint8_t pred) +{ + if (c->nfix == c->fix_cap) { + size_t nc = c->fix_cap ? c->fix_cap * 2 : 8; + void *p = realloc(c->fix, nc * sizeof *c->fix); + + if (!p) { fail(c, "out of memory"); return -1; } + c->fix = p; + c->fix_cap = nc; + } + c->fix[c->nfix].at = (uint32_t)c->n; + c->fix[c->nfix].target = target; + /* The fixup rebuilds the branch from scratch, so the predicate it was + * emitted with has to survive here or a conditional branch is patched + * into an unconditional one - every if taken, every loop endless. */ + c->fix[c->nfix].pred = pred; + c->nfix++; + return 0; +} + +/* Where a sampler's descriptor sits in the secondary bank. The register field + * is seven bits, and the value was truncated to it before the encoder could + * complain - a program with enough uniforms wrapped back into range and read + * three unrelated floats as {ctl, fmt, addr}, which faults or never returns. */ +/* What the frame delivers for a sampler unit, from the options: the class + * of the register the fetch returns and how many planes the surface is + * stored in. Past the caller's array both are the packed 8888 default. */ +static unsigned tex_class_of(const struct cg *c, unsigned unit) +{ + if (c->opts.tex_class && unit < c->opts.ntex_units) + return c->opts.tex_class[unit]; + return UIR_TEXCLASS_U8888; +} + +static unsigned tex_chunks_of(const struct cg *c, unsigned unit) +{ + if (c->opts.tex_chunks && unit < c->opts.ntex_units && + c->opts.tex_chunks[unit] > 1) + return c->opts.tex_chunks[unit]; + return 1; +} + +/* Sampler u's first state block, unit-major over the chunks before it. */ +static uint32_t smp_slot_of(const struct cg *c, unsigned unit) +{ + uint32_t s = 0; + unsigned u; + + for (u = 0; u < unit; u++) + s += tex_chunks_of(c, u); + return s; +} + +static uint32_t smp_slots_total(const struct cg *c) +{ + return smp_slot_of(c, c->nsamp); +} + +/* Where the sample reads its {ctl, fmt, addr} block. This has to be the layout + * alloc_regs() laid down - smp_base, then SMP_SLOT registers a block, a block + * per chunk - and not a formula of its own. A quad per uniform and a quad per + * sampler happened to agree with it for as long as both were true; once + * uniforms were packed to the widths they are read through, the sample went on + * reading where the literal pool now starts and alacritty's glyphs sampled + * constants. */ +static int smp_desc_reg(struct cg *c, uint32_t index, uint32_t chunk, + uint8_t *out) +{ + uint32_t r = c->smp_base + SMP_SLOT * (smp_slot_of(c, index) + chunk); + + if (r > 127u) { + fail(c, "sampler %u chunk %u needs sa%u, past the %u the " + "encoding reaches", index, chunk, r, 127u); + return -1; + } + *out = (uint8_t)r; + return 0; +} + +/* The packed 8888 dword's first byte is blue, the order the pixel back end + * reads: byte k of the texel for channel r, g, b, a. */ +static const uint8_t u8888_byte[4] = { 2, 1, 0, 3 }; + +/* One unpack, emitted straight: byte or half `comp` of src into dst as a + * float, scaled to [0, 1] when the source is normalised. */ +static void emit_unpack(struct cg *c, unsigned op, uint8_t dst, uint8_t sbank, + uint8_t src, unsigned comp, int scale) +{ + struct usse_insn p; + + memset(&p, 0, sizeof p); + p.op = (uint8_t)op; + p.mask = 1; + p.scale = (uint8_t)scale; + p.bytemask = 0xF; /* an f32 is written whole */ + p.dst.bank = USSE_TEMP; + p.dst.num = dst; + p.src[1].bank = sbank; + p.src[1].num = src; + p.src[1].comp = (uint8_t)comp; + p.src[2] = p.src[1]; + emit(c, &p); +} + +static void emit_wdf(struct cg *c, unsigned drc) +{ + struct usse_insn p; + + if (getenv("SGX_NO_WDF")) + return; + memset(&p, 0, sizeof p); + p.op = USSE_WDF; + p.smp_drc = (uint8_t)drc; + emit(c, &p); + /* A probe: how many waits it takes. If one wait retires one + * outstanding read rather than all of them, more of them should help + * in proportion. */ + { + const char *e = getenv("SGX_WDF_COUNT"); + unsigned k, extra = e && *e ? (unsigned)atoi(e) : 0; + + for (k = 1; k < extra; k++) { + memset(&p, 0, sizeof p); + p.op = USSE_WDF; + emit(c, &p); + } + } +} + +/* How many channels a class delivers per chunk. */ +static unsigned tex_class_chans(unsigned cls) +{ + switch (cls) { + case UIR_TEXCLASS_F1616: case UIR_TEXCLASS_U1616: + case UIR_TEXCLASS_S1616: + return 2; + case UIR_TEXCLASS_U8888: + return 4; + default: + return 1; + } +} + +/* Where the chunks' registers go. A float chunk is already the channel it + * stands for, so it lands in the destination directly - chunk k is channel + * k. Anything else is unpacked afterwards, so the raw registers go to + * scratch; a packed 8888 texel keeps the old in-place unpack, which writes + * the quad back to front so the source survives. */ +static struct opnd smp_raw_dest(struct cg *c, struct opnd d, unsigned cls, + unsigned nchunk) +{ + if (cls == UIR_TEXCLASS_F32 || cls == UIR_TEXCLASS_U8888 || + d.bank != USSE_TEMP) + return d; + return scratch_qw(c, 1, nchunk); +} + +/* Four floats out of what the unit delivered. The channels the format + * does not carry read as GL says a texture without them samples: zero, + * and one for alpha. */ +static void smp_unpack_class(struct cg *c, struct opnd d, struct opnd raw, + unsigned cls, unsigned nchunk) +{ + static const uint16_t reset[4] = { + MOE_INC1, MOE_INC1, MOE_INC1, MOE_INC1 + }; + unsigned chans = tex_class_chans(cls) * nchunk, k, op, scale = 1; + + if (chans > QUAD) + chans = QUAD; + moe_need(c, reset); + switch (cls) { + case UIR_TEXCLASS_U8888: + /* Back to front: the texel sits in the first register of the + * quad it unpacks into, so writing that one first would + * destroy the source the other three still read. */ + for (k = QUAD; k-- > 0; ) + emit_unpack(c, USSE_UNPCKF32U8, (uint8_t)(d.base + k), + raw.bank, raw.base, u8888_byte[k], 1); + return; + case UIR_TEXCLASS_F32: + break; + case UIR_TEXCLASS_F16: case UIR_TEXCLASS_F1616: + op = USSE_UNPCKF32F16; scale = 0; + goto halves; + case UIR_TEXCLASS_U16: case UIR_TEXCLASS_U1616: + op = USSE_UNPCKF32U16; + goto halves; + case UIR_TEXCLASS_S16: case UIR_TEXCLASS_S1616: + op = USSE_UNPCKF32S16; +halves: + /* Channel j is half j % per of chunk j / per, low half first. + * The component a pack names is a BYTE offset into the source + * register, so the halves of a 16-bit format are 0 and 2 - + * the vendor's own encoder refuses anything else for U16, S16 + * and F16 (usp/hwinst.c HWInstEncodePCKUNPCKInstNonVec, + * "Byte-offset must be 0 or 2"). Naming them 0 and 1 read the + * low half correctly and the high half as zero: RGBA16F came + * back with red and blue right and green zero. */ + for (k = 0; k < chans; k++) { + unsigned per = tex_class_chans(cls); + + emit_unpack(c, op, (uint8_t)(d.base + k), raw.bank, + (uint8_t)(raw.base + k / per), + 2u * (k % per), (int)scale); + } + break; + default: + fail(c, "texel class %u has no unpack", cls); + return; + } + for (k = chans; k < QUAD; k++) { + struct usse_insn p; + + memset(&p, 0, sizeof p); + p.op = USSE_MOV; + p.mask = 1; + p.dst.bank = USSE_TEMP; + p.dst.num = (uint8_t)(d.base + k); + p.src[1].bank = USSE_CONST; + p.src[1].num = k == 3 ? USSE_C_ONE : USSE_C_ZERO; + emit(c, &p); + } +} + +static void lower_insn_body(struct cg *c, const struct uir_insn *n, size_t idx); + +/* Each instruction claims its scratch afresh, and how much it reaches differs - + * most take one quad where the widest takes three. Recording it per + * instruction lets the placement below ask for that much rather than for the + * worst case, which is the difference between finding a hole and not. */ +static void lower_insn(struct cg *c, const struct uir_insn *n, size_t idx) +{ + unsigned q; + + if (c->scratch_placed && c->s && idx < c->s->ninsns) + for (q = 0; q < SCRATCH_QUADS; q++) + c->scratch_base[q] = + c->scratch_at[idx * SCRATCH_QUADS + q]; + lower_insn_body(c, n, idx); + if (c->scratch_need && c->s && idx < c->s->ninsns) { + c->scratch_need[idx] = c->scratch_nreg; + if (getenv("SGX_SCRATCH_WHO") && c->scratch_nreg >= 8) + fprintf(stderr, "sgx: scratch: insn %u op %u takes %u " + "register(s)\n", (unsigned)idx, n->op, + c->scratch_nreg); + if (c->scratch_w) + for (q = 0; q < SCRATCH_QUADS; q++) + if (c->scratch_qwidth[q] > + c->scratch_w[idx * SCRATCH_QUADS + q]) + c->scratch_w[idx * SCRATCH_QUADS + q] = + c->scratch_qwidth[q]; + } +} + +static void lower_insn_body(struct cg *c, const struct uir_insn *n, size_t idx) +{ + if (c->scratch_at && c->s && idx < c->s->ninsns) + c->scratch = c->scratch_at[idx]; + scratch_reset(c); + materialise_uniforms(c, n); + + struct usse_insn proto; + struct opnd s[3], d, t; + unsigned mask = n->writemask; + + memset(&proto, 0, sizeof proto); + memset(s, 0, sizeof s); + + switch (n->op) { + case UIR_NOP: + case UIR_RET: + return; + + case UIR_LABEL: { + int join = c->moe_old || + (n->label_id <= c->s->next_label && + c->label_target[n->label_id]); + + /* Only a label a branch names is a join. One that nothing + * names is fallen into, so the state the instruction before + * it left still stands - and every shader this front end + * emits ends in such a label. */ + if (join) + moe_block_end(c); + if (n->label_id < c->s->next_label) + c->labelpc[n->label_id] = (uint32_t)c->n; + if (join) + moe_join(c); + return; + } + + case UIR_MOV: + d = resolve(c, &n->dst); + s[1] = resolve(c, &n->src[0]); + emit_vec(c, USSE_MOV, &d, mask, s, 0x2, &proto); + break; + + case UIR_ADD: + case UIR_SUB: + d = resolve(c, &n->dst); + s[0] = legal_src0(c, resolve(c, &n->src[0]), 0, mask); + s[1] = konst(USSE_C_ONE, 0); + s[2] = resolve(c, &n->src[1]); + if (n->op == UIR_SUB) + s[2].neg = !s[2].neg; + emit_vec(c, USSE_MAD, &d, mask, s, 0x7, &proto); + break; + + case UIR_MUL: + d = resolve(c, &n->dst); + s[0] = resolve(c, &n->src[0]); + s[1] = resolve(c, &n->src[1]); + if (s[0].bank != USSE_TEMP && s[0].bank != USSE_PA && + (s[1].bank == USSE_TEMP || s[1].bank == USSE_PA)) { + struct opnd tmp = s[0]; s[0] = s[1]; s[1] = tmp; + } + s[0] = legal_src0(c, s[0], 0, mask); + s[2] = konst(USSE_C_ZERO, 1); + emit_vec(c, USSE_MAD, &d, mask, s, 0x7, &proto); + break; + + case UIR_MAD: + d = resolve(c, &n->dst); + s[0] = resolve(c, &n->src[0]); + s[1] = resolve(c, &n->src[1]); + if (s[0].bank != USSE_TEMP && s[0].bank != USSE_PA && + (s[1].bank == USSE_TEMP || s[1].bank == USSE_PA)) { + struct opnd tmp = s[0]; s[0] = s[1]; s[1] = tmp; + } + s[0] = legal_src0(c, s[0], 0, mask); + s[2] = resolve(c, &n->src[2]); + emit_vec(c, USSE_MAD, &d, mask, s, 0x7, &proto); + break; + + case UIR_MIN: + case UIR_MAX: + d = resolve(c, &n->dst); + s[1] = resolve(c, &n->src[0]); + s[2] = resolve(c, &n->src[1]); + emit_vec(c, n->op == UIR_MIN ? USSE_MIN : USSE_MAX, &d, mask, + s, 0x6, &proto); + break; + + /* Both sources the same register, which is what Imagination's + * compiler emits for dFdx: "fdsx r0, r0, r0". Its own model restricts + * both sources of a gradient to the temporary and primary attribute + * banks (usc2/inst.c g_apfDSXDSYSourceBankRestrictions), so anything + * else is copied first. */ + case UIR_DDX: + case UIR_DDY: + d = resolve(c, &n->dst); + s[1] = s[2] = legal_src0(c, resolve(c, &n->src[0]), 0, mask); + emit_vec(c, n->op == UIR_DDX ? USSE_DSX : USSE_DSY, &d, mask, + s, 0x6, &proto); + break; + + /* The face, the way the vendor's compiler reads it (usc2/icvt_core.c + * CheckFaceType, icvt_f32.c UF_MISC_FACETYPE): a bitwise test of bit 0 + * of the special bank's BFCONTROL register sets a predicate for a + * clockwise triangle, +1.0 is written, and the predicated negate turns + * it into -1.0 where the bit was set. The bitwise form writes its ALU + * result as well as the predicate - the vendor's assembler refuses + * the suppressed form - so it lands in a scratch register. The same + * TEST-then-predicated pair UIR_BRC emits, on p0. */ + case UIR_FACE: { + struct opnd bd = scratch_qw(c, 1, 1); + + d = resolve(c, &n->dst); + memset(&proto, 0, sizeof proto); + proto.op = USSE_TEST; + proto.skipinv = 1; + proto.mask = 1; + proto.test_type = USSE_TEST_TANZ; + proto.test_alu = USSE_TEST_ALU_AND; + proto.test_wr = 1; + proto.test_pdst = 0; /* p0 */ + proto.dst.bank = bd.bank; + proto.dst.num = bd.base; + proto.src[1].bank = USSE_CONST; + proto.src[1].num = USSE_G_BFCONTROL; + proto.src[2].bank = USSE_IMMB; + proto.src[2].num = 1; + emit(c, &proto); + + memset(&proto, 0, sizeof proto); + memset(s, 0, sizeof s); + s[1] = konst(USSE_C_ONE, 0); + emit_vec(c, USSE_MOV, &d, mask, s, 0x2, &proto); + + memset(&proto, 0, sizeof proto); + memset(s, 0, sizeof s); + proto.pred = 1; /* p0: clockwise */ + s[1] = konst(USSE_C_ONE, 1); + s[2] = konst(USSE_C_ONE, 1); + emit_vec(c, USSE_MIN, &d, mask, s, 0x6, &proto); + break; + } + + case UIR_FRC: + d = resolve(c, &n->dst); + s[1] = s[2] = resolve(c, &n->src[0]); + emit_vec(c, USSE_SUBFLR, &d, mask, s, 0x6, &proto); + break; + + case UIR_FLOOR: + d = resolve(c, &n->dst); + t = scratch_q(c, 0); + s[1] = s[2] = resolve(c, &n->src[0]); + emit_vec(c, USSE_SUBFLR, &t, mask, s, 0x6, &proto); + memset(s, 0, sizeof s); + s[0] = legal_src0(c, resolve(c, &n->src[0]), 1, mask); + s[1] = konst(USSE_C_ONE, 0); + s[2] = t; + s[2].neg = 1; + emit_vec(c, USSE_MAD, &d, mask, s, 0x7, &proto); + break; + + case UIR_RCP: + case UIR_RSQ: + case UIR_LOG2: + case UIR_EXP2: { + unsigned uop = n->op == UIR_RCP ? USSE_RCP : + n->op == UIR_RSQ ? USSE_RSQ : + n->op == UIR_LOG2 ? USSE_LOG : USSE_EXP; + + d = resolve(c, &n->dst); + s[1] = resolve(c, &n->src[0]); + emit_vec(c, uop, &d, mask, s, 0x2, &proto); + break; + } + + case UIR_SQRT: + d = resolve(c, &n->dst); + t = scratch_q(c, 0); + s[1] = resolve(c, &n->src[0]); + emit_vec(c, USSE_RSQ, &t, mask, s, 0x2, &proto); + memset(s, 0, sizeof s); + s[1] = t; + emit_vec(c, USSE_RCP, &d, mask, s, 0x2, &proto); + break; + + case UIR_DP3: + lower_dot(c, n, 3); + break; + case UIR_DP4: + lower_dot(c, n, 4); + break; + + case UIR_NRM: { + struct uir_insn dot = *n; + + d = resolve(c, &n->dst); + dot.src[1] = dot.src[0]; + dot.dst.cls = UIR_REG_VIRT; + dot.writemask = 1; + /* dot3 of the source with itself, into scratch quad 1 */ + { + struct usse_insn p2; + struct opnd ds[3], sc = scratch_q(c, 1); + uint16_t want[4]; + unsigned i; + + memset(&p2, 0, sizeof p2); + memset(ds, 0, sizeof ds); + ds[1] = ds[2] = resolve(c, &n->src[0]); + want[0] = MOE_INC0; + want[1] = MOE_INC1; + for (i = 0; i < 2; i++) + want[i + 2] = ds[i + 1].swz == UIR_SWZ_XYZW + ? MOE_INC1 : MOE_SWZ(ds[i + 1].swz); + moe_need(c, want); + p2.op = USSE_DP; + p2.repeat = 3; + p2.dst.bank = sc.bank; p2.dst.num = sc.base; + p2.src[1].bank = ds[1].bank; p2.src[1].num = ds[1].base; + p2.src[2].bank = ds[2].bank; p2.src[2].num = ds[2].base; + emit(c, &p2); + + memset(&p2, 0, sizeof p2); + memset(ds, 0, sizeof ds); + ds[1] = sc; + emit_vec(c, USSE_RSQ, &sc, 1, ds, 0x2, &p2); + + memset(&p2, 0, sizeof p2); + memset(ds, 0, sizeof ds); + ds[0] = legal_src0(c, resolve(c, &n->src[0]), 0, mask); + ds[1] = sc; + ds[1].swz = UIR_SWZ_XXXX; + ds[2] = konst(USSE_C_ZERO, 1); + emit_vec(c, USSE_MAD, &d, mask, ds, 0x7, &p2); + } + break; + } + + case UIR_CROSS: { + static const unsigned ai[3] = { 1, 2, 0 }, bi[3] = { 2, 0, 1 }; + struct opnd a = resolve(c, &n->src[0]), b = resolve(c, &n->src[1]); + unsigned k; + + d = resolve(c, &n->dst); + t = scratch_q(c, 0); + for (k = 0; k < 3; k++) { + struct usse_insn p2; + struct opnd cs[3]; + unsigned dm = 1u << k; + + if (!(mask & dm)) + continue; + memset(&p2, 0, sizeof p2); + memset(cs, 0, sizeof cs); + /* t.k = a[ai] * b[bi] */ + cs[0] = a; cs[0].swz = (uint8_t)(chan(a.swz, ai[k]) * 0x55); + cs[0] = legal_src0(c, cs[0], 1, dm); + cs[1] = b; cs[1].swz = (uint8_t)(chan(b.swz, bi[k]) * 0x55); + cs[2] = konst(USSE_C_ZERO, 1); + emit_vec(c, USSE_MAD, &t, dm, cs, 0x7, &p2); + + memset(&p2, 0, sizeof p2); + memset(cs, 0, sizeof cs); + /* d.k = -a[bi] * b[ai] + t.k */ + cs[0] = a; cs[0].swz = (uint8_t)(chan(a.swz, bi[k]) * 0x55); + cs[0] = legal_src0(c, cs[0], 1, dm); + cs[0].neg = !cs[0].neg; + cs[1] = b; cs[1].swz = (uint8_t)(chan(b.swz, ai[k]) * 0x55); + cs[2] = t; + emit_vec(c, USSE_MAD, &d, dm, cs, 0x7, &p2); + } + break; + } + + case UIR_SLT: lower_cmpset(c, n, USSE_MOVC_TN, 0); break; + case UIR_SGE: lower_cmpset(c, n, USSE_MOVC_TN, 1); break; + case UIR_SEQ: lower_cmpset(c, n, USSE_MOVC_TZ, 0); break; + case UIR_SNE: lower_cmpset(c, n, USSE_MOVC_TNZ, 0); break; + + case UIR_CMP: + d = resolve(c, &n->dst); + proto.movc_test = USSE_MOVC_TN; + s[1] = flatten(c, resolve(c, &n->src[1]), 1, mask); + s[2] = flatten(c, resolve(c, &n->src[2]), 2, mask); + /* movc has no modifiers on the condition either */ + s[0] = legal_src0(c, flatten(c, resolve(c, &n->src[0]), 3, mask), + 0, mask); + emit_vec(c, USSE_MOVC, &d, mask, s, 0x7, &proto); + break; + + case UIR_TEX: + case UIR_TEXPROJ: + case UIR_TEXLOD: + case UIR_TEXBIAS: { + struct opnd coord = resolve(c, &n->src[0]); + struct opnd lod; + unsigned unit, cls, nchunk, k; + uint8_t lodm = n->op == UIR_TEXLOD ? USSE_SMP_LODM_REPLACE : + n->op == UIR_TEXBIAS ? USSE_SMP_LODM_BIAS : + USSE_SMP_LODM_NONE; + /* Where each chunk's register lands before it is unpacked. */ + struct opnd raw; + + if (n->src[1].cls != UIR_REG_SAMPLER) { + fail(c, "instruction %zu: tex needs a sampler operand", idx); + return; + } + unit = n->src[1].index; + cls = tex_class_of(c, unit); + nchunk = tex_chunks_of(c, unit); + d = resolve(c, &n->dst); + memset(&lod, 0, sizeof lod); + + /* Delivered rather than sampled: the texel of unit u is + * already in pa[1 + u], one register, in the render target's + * packed form. */ + if (c->fragment && c->frag_tex_preiterated) { + struct opnd src[3]; + struct usse_insn p2; + + /* A projected one is delivered too. The record + * carries the coordinate already divided, so what + * the unit iterates is the projected coordinate and + * the texel it fetches is the projected sample. */ + if (n->op != UIR_TEX && n->op != UIR_TEXPROJ) { + fail(c, "instruction %zu: only a sample is " + "delivered, plain or projected", idx); + return; + } + if (n->src[0].cls != UIR_REG_IN && + n->src[0].cls != UIR_REG_TEXCOORD) { + fail(c, "instruction %zu: the coordinate is " + "the iterated one, so a computed " + "coordinate cannot be sampled", idx); + return; + } + /* The iterator hands over one packed dword; a float + * or 16-bit texel, or one in several planes, is only + * reachable by sampling for it. */ + if (cls != UIR_TEXCLASS_U8888 || nchunk > 1) { + fail(c, "instruction %zu: the iterator delivers " + "a packed 8888 texel, so unit %u's " + "format samples for itself", idx, unit); + return; + } + memset(src, 0, sizeof src); + memset(&p2, 0, sizeof p2); + /* A move takes its source in slot 1. */ + src[1].bank = USSE_PA; + src[1].base = (uint8_t)(c->frag_tex_pa_base + + n->src[1].index); + /* A probe. Where the primary emitter puts the texel + * of a second sampled set is not established - see + * work/attrib-binding README 13 - and the two are not + * contiguous, so sweeping the base alone cannot find + * the second. This names it on its own. */ + if (n->src[1].index == 1) { + const char *e = getenv("SGX_TEX_PA1"); + + if (e && *e) + src[1].base = (uint8_t)atoi(e); + } + src[1].swz = UIR_SWZ_XYZW; + if (c->frag_out_packed) { + emit_vec(c, USSE_MOV, &d, 1, src, 0x2, &p2); + break; + } + /* The texel arrives as one packed 8888 dword, which a + * program that only moves it to the output can hand + * over as it stands. One that computes with it needs + * four floats, so each byte is unpacked into its own + * register - and the dword's first byte is blue, the + * order the pixel back end reads. */ + { + static const uint16_t reset[4] = { + MOE_INC1, MOE_INC1, MOE_INC1, MOE_INC1 + }; + + /* Emitted straight, so the MOE has to be put + * back first or a swizzle left by the last + * vector instruction redirects the source. */ + moe_need(c, reset); + for (k = 0; k < QUAD; k++) + emit_unpack(c, USSE_UNPCKF32U8, + (uint8_t)(d.base + k), + USSE_PA, src[1].base, + u8888_byte[k], 1); + } + break; + } + if (d.bank != USSE_TEMP && d.bank != USSE_PA) { + fail(c, "instruction %zu: sample destination must be a " + "temporary", idx); + return; + } + /* The unit's result is in the surface's own form, and only + * the packed 8888 one is what the pixel back end reads. */ + if (c->frag_out_packed && cls != UIR_TEXCLASS_U8888) { + fail(c, "instruction %zu: unit %u's texel is not a " + "packed 8888 dword, so the program cannot end " + "in a move of it", idx, unit); + return; + } + /* The LOD operand: one float, read from the x the swizzle + * names. The sample form has no modifiers and the constant + * banks are not reachable from its third slot, so those are + * copied to scratch. */ + if (lodm) { + struct opnd l = resolve(c, &n->src[2]); + + l.base = (uint8_t)(l.base + chan(l.swz, 0)); + l.swz = UIR_SWZ_XYZW; + if (l.neg || l.abs || l.bank == USSE_CONST || + l.bank == USSE_IMMB) + l = flatten_w(c, l, 0, 1, 1); + lod = l; + } + /* The coordinate comes from the iterator. + * + * Measured on hardware (work/smp-reverse/README.md section 4): + * SMP reads its coordinate from the primary-attribute bank as + * an f32 pair and does not consult the register file at all. + * A coordinate handed over in a temporary - packed to f16 or + * not - leaves the sampler at its default of texel (0,0), so + * every pixel comes back the same and a textured draw is one + * flat colour. That is what ioquake3's menu rendered as. + * + * So when the coordinate is an input the iterator delivered, + * name that register and sample f32. A projected sample needs + * no arithmetic here either: the coordinate set carries the + * divisor and the iterator applies it, which is what the + * three-floats-projected set encoding is for. */ + if (coord.bank == USSE_PA && !getenv("SGX_SMP_F16_COORD")) { + static const uint16_t reset[4] = { + MOE_INC1, MOE_INC1, MOE_INC1, MOE_INC1 + }; + + /* Emitted straight, like the packs: a swizzle the last + * vector instruction left on a source slot moves this + * sample's coordinate or state register by that + * swizzle's first channel. The f16 path resets the MOE + * before its pack; this path went out under whatever + * was in force. */ + moe_need(c, reset); + memset(&proto, 0, sizeof proto); + proto.op = USSE_SMP; + /* CDIM: a volume or cube reads a third float, the + * next register on from the pair - USE1 [11:10], + * EURASIA_USE1_SMP_CDIM_UVS. The set was iterated + * three wide for it (sgx_shader.c, volume). */ + proto.smp_dim = n->tex_dim == 3 ? 3 : 2; + proto.smp_drc = 0; + proto.smp_f16 = 0; + proto.smp_lodm = lodm; + proto.mask = 1; + proto.src[0].bank = USSE_PA; + /* A vector is consecutive registers, one component + * each - mov.repeat4 r0, pa0 fills r0..r3 - so the + * coordinate pair starts at the register holding its + * first component, not at the varying's base. Mesa + * packs two varyings into one, so a coordinate sitting + * at .yz is the ordinary case. */ + proto.src[0].num = (uint8_t)(coord.base + + (coord.swz & 3u)); + proto.src[1].bank = USSE_SA; + proto.src[2].bank = lod.bank; + proto.src[2].num = lod.base; + raw = smp_raw_dest(c, d, cls, nchunk); + for (k = 0; k < nchunk; k++) { + proto.dst.bank = raw.bank; + proto.dst.num = (uint8_t)(raw.base + k); + if (smp_desc_reg(c, unit, k, &proto.src[1].num)) + return; + emit(c, &proto); + emit_wdf(c, 0); + } + goto smp_unpack; + } + /* A third component had nowhere to go while the coordinate + * was packed into one register as an f16 pair. Delivered as + * f32 it is simply the next register, so a computed volume + * or cube coordinate is no longer a refusal here. */ + if (n->op == UIR_TEXPROJ) { + struct usse_insn p2; + struct opnd ps[3], qd, q = scratch_q(c, 1); + + memset(&p2, 0, sizeof p2); + memset(ps, 0, sizeof ps); + /* The coordinate's own swizzle composed, not replaced. + * Overwriting it made the reciprocal read the fourth + * register of the operand rather than whichever + * component the swizzle puts in w - and the multiply + * below does compose, so the two halves of the divide + * disagreed. Mesa packs two varyings into one input, + * so a coordinate reading IN[0].yzxx is the normal + * case, not a corner one. */ + { + unsigned wc = getenv("SGX_TXP_OLD_SWZ") ? 3u : + ((coord.swz >> 6) & 3u); + + ps[1] = coord; + ps[1].swz = (unsigned char)(wc | (wc << 2) | + (wc << 4) | + (wc << 6)); + } + emit_vec(c, USSE_RCP, &q, 1, ps, 0x2, &p2); + memset(&p2, 0, sizeof p2); + memset(ps, 0, sizeof ps); + ps[0] = legal_src0(c, coord, 0, 3); + ps[1] = q; + ps[1].swz = UIR_SWZ_XXXX; + ps[2] = konst(USSE_C_ZERO, 1); + /* Not back into q: the multiply iterates over the + * components while reading the reciprocal from q.x, so + * writing the first result there left every later + * component multiplied by the first quotient instead + * of by 1/w. */ + qd = scratch_q(c, 3); + emit_vec(c, USSE_MAD, &qd, 3, ps, 0x7, &p2); + coord = qd; + } + /* The sample form reads consecutive registers and carries no + * source modifiers, so anything else is copied to scratch - + * as many as the target has components, since that is what + * the unit reads from src0 on. */ + { + unsigned nc = n->tex_dim == 3 ? 3u : 2u; + + coord = flatten_w(c, coord, 3, (1u << nc) - 1u, nc); + } + if (coord.bank == USSE_CONST || coord.bank == USSE_IMMB) { + fail(c, "instruction %zu: sample coordinate bank not " + "addressable", idx); + return; + } + { + uint16_t want[4] = { MOE_INC1, MOE_INC1, MOE_INC1, MOE_INC1 }; + unsigned nc = n->tex_dim == 3 ? 3u : 2u; + /* The packed form needs a register for the pair, and + * one more for the LOD beside it; the f32 form reads + * the registers the coordinate is already in. */ + int packed = getenv("SGX_SMP_F16_COORD") != NULL; + struct opnd cf = packed ? + scratch_qw(c, 2, lodm ? 2 : 1) : coord; + + moe_need(c, want); + /* The coordinate goes to the unit as two (or three) + * f32 registers, which is the form the SMP's CTYPE + * field calls F32 - EURASIA_USE1_SMP_CTYPE_F32 is 0, + * sgxdefs.h:6348 - and the form the vendor's own + * compiler emits: usc2's ConvertTextureSampleToUseasm + * copies the coordinate register as it stands and tags + * a format only when it is F16 or C10 (asm.c:3394). + * + * This used to pack the pair into one register and + * sample it as an f16 pair, on the strength of a + * measurement that f32 registers "sampled a coordinate + * built from their bit patterns - the low half of an + * f32 whose fraction is exact is zero, so one axis was + * pinned to zero". That is what an F16 coordinate type + * reading two f32 registers does, and it is what the + * run that produced it had: the registers changed and + * the type did not. Sampling them with the type that + * matches costs one instruction and one register less. + * SGX_SMP_F16_COORD puts the packed form back. */ + if (packed) { + memset(&proto, 0, sizeof proto); + proto.op = USSE_PCKF16F32; + proto.mask = 1; + proto.bytemask = 0xF; + proto.dst.bank = USSE_TEMP; + proto.dst.num = (uint8_t)cf.base; + proto.src[1].bank = coord.bank; + proto.src[1].num = coord.base; + proto.src[2].bank = coord.bank; + proto.src[2].num = (uint8_t)(coord.base + 1); + emit(c, &proto); + /* The LOD shares the coordinate's type (usc2 + * asm.c GetTextureSampleSourceFormat reads it + * from either), so it is packed too. */ + if (lodm) { + memset(&proto, 0, sizeof proto); + proto.op = USSE_PCKF16F32; + proto.mask = 1; + proto.bytemask = 0x3; + proto.dst.bank = USSE_TEMP; + proto.dst.num = (uint8_t)(cf.base + 1); + proto.src[1].bank = lod.bank; + proto.src[1].num = lod.base; + proto.src[2] = proto.src[1]; + emit(c, &proto); + } + c->coord_f16 = 1; + } + + memset(&proto, 0, sizeof proto); + proto.op = USSE_SMP; + proto.smp_dim = (uint8_t)nc; + proto.smp_drc = getenv("SGX_DRC_PER_SAMPLER") ? + (unit & 1u) : 0u; + proto.smp_f16 = packed ? 1 : 0; + proto.smp_lodm = lodm; + proto.mask = 1; + proto.src[0].bank = cf.bank; + proto.src[0].num = cf.base; + if (packed) { + proto.src[2].bank = USSE_TEMP; + proto.src[2].num = (uint8_t)(cf.base + 1); + } else { + proto.src[2].bank = lod.bank; + proto.src[2].num = lod.base; + } + proto.src[1].bank = USSE_SA; + raw = smp_raw_dest(c, d, cls, nchunk); + for (k = 0; k < nchunk; k++) { + proto.dst.bank = raw.bank; + proto.dst.num = (uint8_t)(raw.base + k); + if (smp_desc_reg(c, unit, k, &proto.src[1].num)) + return; + /* No .syncs. SyncStart on a sample forces a + * deschedule the task never resumes from: the + * read never completes, the wait on it never + * returns, and the render stalls with no fault + * to explain it. The vendor's own output + * carries neither .syncs nor .skipinv on a + * sample. */ + emit(c, &proto); + /* The sample is a dependent read: without + * waiting on it the program reads the texel + * register before the texture unit has written + * it. The vendor emits one of these after + * every sample. */ + emit_wdf(c, proto.smp_drc); + } + } + +smp_unpack: + /* What lands is one register per chunk holding the texel in + * the surface's own form, not four floats. A program that + * computes with it needs the channels as floats. */ + if (!c->frag_out_packed && d.bank == USSE_TEMP) + smp_unpack_class(c, d, raw, cls, nchunk); + break; + } + + case UIR_KILL: { + struct opnd cv, lo; + + + /* The condition is kept in a register and applied at the end, + * where a movc puts the destination back over any fragment + * whose condition went negative. That needs the destination in + * o0, which the read-modify-write object type provides - the + * result reports it so the driver can select it. + * + * Depth is still written for a discarded fragment, which GLSL + * forbids; a depth-correct discard needs ATST8. */ + cv = flatten(c, resolve(c, &n->src[0]), 1, 0xF); + + /* A discard happens when any channel goes negative, so the + * four reduce to their minimum and one condition carries all + * of them. */ + memset(&proto, 0, sizeof proto); + /* Two registers, not a quad: the reduction writes x and y and + * the next step reads y. Asking for a quad here and one below + * made a discard the widest scratch user in the program - + * twelve registers where seven are touched - and that is what + * put alacritty's fragment shader past the count a task can + * name. */ + lo = scratch_qw(c, 2, 2); + s[1] = cv; + s[2] = cv; + s[2].swz = UIR_SWZ(2, 3, 2, 3); + emit_vec(c, USSE_MIN, &lo, 0x3, s, 0x6, &proto); + + memset(&proto, 0, sizeof proto); + d = scratch_qw(c, 3, 1); + s[1] = lo; + s[2] = lo; + s[2].swz = UIR_SWZ(1, 1, 1, 1); + emit_vec(c, USSE_MIN, &d, 0x1, s, 0x6, &proto); + + /* The movc that applies this tests the sign bit, and a + * comparison that came out false and was then negated is + * -0.0 - which is not less than zero, but has the bit set. + * Adding zero turns it back into +0.0 and leaves every other + * value alone. */ + memset(&proto, 0, sizeof proto); + memset(s, 0, sizeof s); + s[0] = d; + s[1] = konst(USSE_C_ONE, 0); + s[2] = konst(USSE_C_ZERO, 0); + d = temp(c->killreg); + emit_vec(c, USSE_MAD, &d, 0x1, s, 0x7, &proto); + + /* The vendor's kill test: adding zero and asking whether the + * result is positive or zero leaves "this fragment survives" + * in p1, which the closing write is predicated on. The vector + * write is suppressed, so the destination is a placeholder. + * A second kill predicates its own test on p1. */ + memset(&proto, 0, sizeof proto); + proto.op = USSE_TEST; + proto.skipinv = 1; + proto.mask = 1; + proto.test_type = USSE_TEST_POZ; + proto.test_alu = USSE_TEST_FADD; + proto.test_pdst = SGX_PRED_SURVIVE - 1; + proto.pred = c->killed ? SGX_PRED_SURVIVE : 0; + /* The vector write is suppressed, so the destination is a + * placeholder - but aim it at the register the condition is + * already in, so that a write which is not in fact suppressed + * lands somewhere dead rather than on live state. */ + proto.dst.bank = USSE_TEMP; + proto.dst.num = (uint8_t)c->killreg; + proto.src[1].bank = USSE_TEMP; + proto.src[1].num = (uint8_t)c->killreg; + proto.src[2].bank = USSE_CONST; + proto.src[2].num = USSE_C_ZERO; + emit(c, &proto); + c->killed = 1; + break; + } + + /* A conditional branch is two instructions: a TEST sets a predicate + * from an ALU result, and the branch carries that predicate. Adding + * zero leaves the condition alone, so "does it differ from zero" is + * the whole test - which is what UIR_BRC asks. The forms were read + * back from Imagination's own disassembler; see USSE_TEST_TANZ. */ + case UIR_BRC: { + /* One register: the test writes a single component and the + * branch reads the predicate, not this. */ + struct opnd bd = scratch_qw(c, 1, 1); + + /* The TEST carries no source modifiers, so the condition is + * copied flat first. */ + memset(s, 0, sizeof s); + s[1] = flatten(c, resolve(c, &n->src[0]), 0, 0xF); + s[2] = konst(USSE_C_ZERO, 0); + memset(&proto, 0, sizeof proto); + proto.skipinv = 1; + proto.test_type = USSE_TEST_TANZ; + proto.test_alu = USSE_TEST_ALU_FADD; + proto.test_chan = 0; + proto.test_pdst = 0; /* p0 */ + /* The vector write is suppressed, so the destination is only + * a placeholder the encoding still needs. */ + emit_vec(c, USSE_TEST, &bd, 0x1, s, 0x6, &proto); + /* Before the fixup is recorded: a reset emitted after it + * would take the address the branch was to have. */ + moe_block_end(c); + if (branch_fixup(c, n->branch_target, 1)) + return; + memset(&proto, 0, sizeof proto); + proto.op = USSE_BR; + proto.skipinv = 1; + proto.pred = 1; /* p0 */ + emit(c, &proto); + break; + } + + case UIR_BR: + moe_block_end(c); + if (branch_fixup(c, n->branch_target, 0)) + return; + memset(&proto, 0, sizeof proto); + proto.op = USSE_BR; + emit(c, &proto); + break; + + case UIR_EMIT: + if (c->fragment && c->frag_out_packed) { + emit_frag_tail(c); + } else if (c->fragment) { + struct pkstep { unsigned bm, s1, s2; }; + static const struct pkstep pk_rgb[2] = { + { 0x6, 1, 0 }, { 0x9, 2, 3 } + }; + /* The step that reads component 0 goes first. Both + * packs write the colour's first register, which is + * component 0's own, so a step reading it after the + * other has written must read a destroyed float: it + * put a corrupted byte 0 - the pixel's blue - whose + * value followed red and green. pk_rgb reads it in + * its first step already. */ + static const struct pkstep pk_bgr[2] = { + { 0x9, 0, 3 }, { 0x6, 1, 2 } + }; + /* Which byte of the pixel each channel goes in. + * + * SGX_BGRA swaps them. It exists because glmark2's + * shading scene looked like a red model rendered + * blue - which it is not: that scene's material is + * vec4(0, 0, 1, 1), the model is blue, and the word + * glmark2 prints is ABGR rather than ARGB. There is + * no channel-order defect. The knob is kept only + * because the next person to misread a packed word + * will want to rule it out in one run; the default is + * correct and the feature suite fails 12 of 26 with + * it set. */ + const struct pkstep *pk = + (c->opts.frag_out_reverse || + getenv("SGX_BGRA")) ? pk_bgr : pk_rgb; + static const uint16_t reset[4] = { + MOE_INC1, MOE_INC1, MOE_INC1, MOE_INC1 + }; + unsigned k; + + /* The pair below is emitted straight, so it runs under + * whatever the last vector instruction left in the + * MOE. A source swizzle there is applied to the pack's + * own operands: with the colour's yxyy still in force + * the second pack read one register past the one it + * was given, and the pixel came out white or black + * rather than the colour. */ + moe_need(c, reset); + + /* The pair packs into a register, and one move puts it + * in the pixel. + * + * Packing straight into o0 costs an instruction less + * and is what this did, but it leaves a program whose + * last instruction is a pack. Blending replaces the + * closing instruction with the blend, which reads the + * destination out of o0 - and with two packs writing + * o0, the surviving one destroyed that destination one + * instruction before the blend read it. The vendor + * never lets a blended program write o0 before the + * end; its move into o0 becomes a move into a temp the + * moment blending is enabled. + * + * The pack destination is the colour's own first + * register: the second pack reads +2 and +3, which + * this does not disturb, and each pack reads its + * sources before writing. */ + /* The second colour first, and packed the same way: + * a dual source blend reads it as a pixel beside the + * destination, not as the four floats the program + * computed. Left unpacked it is a float bit pattern + * read as four bytes, which is why every factor that + * named it came out as nothing at all. */ + if (c->has_fragout1) + for (k = 0; k < 2; k++) { + memset(&proto, 0, sizeof proto); + proto.op = USSE_PCKU8F32; + proto.mask = 1; + proto.scale = 1; + proto.bytemask = (uint8_t)pk[k].bm; + proto.dst.bank = USSE_TEMP; + proto.dst.num = (uint8_t)c->fragout1; + proto.src[1].bank = USSE_TEMP; + proto.src[1].num = + (uint8_t)(c->fragout1 + pk[k].s1); + proto.src[2].bank = USSE_TEMP; + proto.src[2].num = + (uint8_t)(c->fragout1 + pk[k].s2); + emit(c, &proto); + } + for (k = 0; k < 2; k++) { + memset(&proto, 0, sizeof proto); + proto.op = USSE_PCKU8F32; + proto.mask = 1; + proto.scale = 1; + proto.bytemask = (uint8_t)pk[k].bm; + proto.dst.bank = USSE_TEMP; + proto.dst.num = (uint8_t)c->fragout; + proto.src[1].bank = USSE_TEMP; + proto.src[1].num = (uint8_t)(c->fragout + pk[k].s1); + proto.src[2].bank = USSE_TEMP; + proto.src[2].num = (uint8_t)(c->fragout + pk[k].s2); + emit(c, &proto); + } + emit_frag_tail(c); + } else { + unsigned sl, top = 0; + + /* Fill the gaps first. Outputs land at o[4 * slot] and + * the slots a program uses need not be contiguous: one + * writing a position and a generic leaves the colour + * slot between them untouched, and the MTE emits it + * anyway. Left unwritten it is whatever the output + * bank last held, which the fragment task then reads + * as a varying. */ + for (sl = 0; sl < 32; sl++) + if (c->out_slot_written & (1u << sl)) + top = sl; + /* Up to what the part emits, not up to the highest + * slot the program happens to write. The gaps below + * that were filled and the slots above it were not, + * so a program writing a position and a colour and + * nothing else - which is every fixed-function + * transform - left the two slots behind them holding + * whatever the output bank had, and the part emitted + * them: the tiling engine stopped mid-scene with no + * MMU fault and the frame was lost. Intermittently, + * because it depends on what the last program left + * there. */ + top = top + 1u; + if (c->opts.vtx_out_slots_emitted > top) + top = c->opts.vtx_out_slots_emitted; + if (top > 32u) + top = 32u; + for (sl = 0; sl < top; sl++) { + static const uint16_t reset[4] = { + MOE_INC1, MOE_INC1, MOE_INC1, MOE_INC1 + }; + + if (c->out_slot_written & (1u << sl)) + continue; + /* Emitted straight, so a swizzle the last + * vector instruction left in the MOE would + * read these constants one register apart. */ + moe_need(c, reset); + memset(&proto, 0, sizeof proto); + proto.op = USSE_MOV; + proto.mask = 0xf; + proto.repeat = QUAD; + proto.dst.bank = USSE_OUT; + proto.dst.num = (uint8_t)(QUAD * sl); + proto.src[1].bank = USSE_CONST; + proto.src[1].num = USSE_C_ZERO; + emit(c, &proto); + } + /* The record's colour slot is four floats whichever + * stage fills it: the MTE reads the base colour as F32 + * x4 (sgxdefs.h EURASIA_MTE_OUTPUT_OFFSET_BASECOLOUR + * to _OFFSETCOLOUR is sixteen bytes; fftnlgles.c adds + * four to the vertex size for EURASIA_MTE_BASE) and it + * is the iterator that packs them into the one 8888 + * dword a packed-colour issue delivers. The draw module + * writes them blue first, and the fragment unpack reads + * that order, so a transform run on the part has to + * write the same: swap red and blue, and leave the four + * floats. Packing them here put an 8888 dword where + * the MTE reads the first float - a colour with alpha + * one is a huge or NaN red - which is a draw that samples + * and reads its colour rendering nothing on the part. */ + if (c->opts.vtx_pack_colour && + (c->out_slot_written & (1u << 1))) { + struct opnd t = scratch_qw(c, 0, 1), s[3], r, b; + + memset(s, 0, sizeof s); + memset(&r, 0, sizeof r); + r.bank = USSE_OUT; + r.base = QUAD; + r.swz = UIR_SWZ_XYZW; + b = r; + b.base = QUAD + 2; + memset(&proto, 0, sizeof proto); + s[1] = b; + emit_vec(c, USSE_MOV, &t, 1, s, 0x2, &proto); + s[1] = r; + emit_vec(c, USSE_MOV, &b, 1, s, 0x2, &proto); + s[1] = t; + emit_vec(c, USSE_MOV, &r, 1, s, 0x2, &proto); + } + memset(&proto, 0, sizeof proto); + proto.op = USSE_EMIT; + proto.emit_target = USSE_EMIT_VTX; + proto.emit_freep = 1; + proto.end = 1; + emit(c, &proto); + } + break; + + default: + fail(c, "instruction %zu: opcode %d has no lowering", idx, (int)n->op); + return; + } + + if (n->saturate && n->op != UIR_EMIT && n->op != UIR_KILL) { + struct opnd sd = resolve(c, &n->dst); + + lower_saturate(c, &sd, mask); + } +} + + +/* Put everything a lowering pass produced back, so the next one starts from + * the same state the first did. */ +static void reset_emit(struct cg *c) +{ + uint32_t i; + + if (c->eff) + fprintf(stderr, "--- pass ---\n"); + c->n = 0; + c->last_op = 0; + c->pairfix = 0; + c->nfix = 0; + /* The pool is not cleared: scan() built it from the shader, and it is + * the same for every emit pass. Rebuilding it here would drop the + * spill marks alloc_regs() made against these very indices. */ + c->nlimm = 0; + c->ntemps = 0; + c->npa = 0; + c->nsa = 0; + c->killed = 0; + c->uni_used = 0; + c->scratch_used = 0; + scratch_reset(c); + c->moe_known = 1; + memset(c->moe, 0, sizeof c->moe); + memset(c->uni_live, 0, sizeof c->uni_live); + memset(c->imm_live, 0, sizeof c->imm_live); + for (i = 0; i < c->nvirt; i++) + c->vbase[i] = NOREG; + if (c->labelpc) + memset(c->labelpc, 0xff, + (size_t)(c->s->next_label + 1) * sizeof *c->labelpc); + if (c->lay) { + c->lay->nmap = 0; + c->lay->npool = 0; + c->lay->nunlowered = 0; + c->lay->unlowered[0] = 0; + } +} + +/* --- EFO formation ------------------------------------------------------ + * + * One EFO carries two multiplies and two adders. Its unified-store + * destination takes one of the four results and the other two go to the + * internal registers i0 and i1 (sgxdefs.h:5398-5427), so a pair of + * independent instructions becomes one instruction plus one internal + * register that a later instruction reads. + * + * What that costs is the internal registers' lifetime. Execution-thread + * switching happens between instruction pairs and loses i0..i3 + * (usc2/finalise.c:325-333), a WDF or an instruction with .syncs forces a + * switch after its own pair, and the internal registers are never live + * across a basic-block boundary. The .nosched flag on either instruction of + * a pair disables descheduling at the end of the *following* pair + * (finalise.c:334-341), which is why the flag has to go two pairs before + * the boundary it protects: usc2 asserts exactly that placement + * (finalise.c:643 psLastInPair == psInst->psPrev->psPrev->psPrev). + * + * The pass runs on the finished stream because that is where the register + * allocation, the MOE state and the instruction addresses are final. Each + * candidate is applied to a trial copy and kept only if the copy still + * encodes, still splits the branch pairs the way emit() does, and can carry + * every .nosched the internal-register lifetime asks for. Nothing is + * assumed: a fusion that cannot be made legal is dropped, not patched over. + */ + +#define EFO_WINDOW 8u /* instructions a partner may be away */ + +/* Why a candidate pair was turned down. SGX_EFO_TRACE prints the tally, + * which is how the reach of the pass on a given shader is measured rather + * than guessed. */ +enum efo_why { + EFO_OK = 0, + EFO_NOT_ALU, /* not two single-iteration multiply-adds */ + EFO_NO_SHAPE, /* no 535 wiring computes both results */ + EFO_DEPEND, /* one feeds the other, or something between */ + EFO_NOT_DEAD, /* the second result is not read exactly once */ + EFO_READER, /* the reader cannot name an internal register */ + EFO_SCHED, /* the lifetime cannot be held across a pair */ + EFO_PAIR, /* it would put two branches in one pair */ + EFO_ENCODE, + EFO_WHY_COUNT +}; + +static const char *const efo_why_name[EFO_WHY_COUNT] = { + "formed", "not-alu", "no-shape", "depend", "not-dead", "reader", + "nosched", "pair", "encode" +}; + +/* One instruction's view as the pass needs it: what it computes, in three + * shapes, since the backend lowers MUL and ADD to FMAD as well as MAD. */ +enum efo_kind { EFO_NONE = 0, EFO_MUL, EFO_ADD, EFO_MAD }; + +struct efo_val { + enum efo_kind kind; + struct usse_operand d; /* destination register */ + struct usse_operand a, b, c; /* a*b + c, a*b, or a + b */ +}; + +/* The banks an EFO operand slot can name. Source 0 has no bank extension + * and sources 1 and 2 have none either (useasm.c:7994 EncodeSrc0/1/2 with + * the extension disabled), so the constant bank is out of reach in all + * three - which is why only the shapes whose constants fall away can fuse. */ +static int efo_bank_ok(const struct usse_operand *o, unsigned slot) +{ + if (slot == 0) + return o->bank == USSE_TEMP || o->bank == USSE_PA; + return o->bank == USSE_TEMP || o->bank == USSE_PA || + o->bank == USSE_OUT || o->bank == USSE_SA; +} + +static int efo_same(const struct usse_operand *x, const struct usse_operand *y) +{ + return x->bank == y->bank && x->num == y->num && x->neg == y->neg && + x->abs == y->abs; +} + +static int is_const(const struct usse_operand *o, unsigned base) +{ + return o->bank == USSE_CONST && o->num >= base && o->num < base + 4; +} + +/* Classify one emitted instruction. Only a single-iteration FMAD under a + * MOE with no swizzled slot qualifies: an EFO's second result lands in one + * internal register, which no repeat steps, and a swizzled slot moves the + * registers an operand names even for one iteration. */ +static int efo_classify(const struct usse_insn *in, const uint16_t moe[4], + struct efo_val *v) +{ + unsigned i; + + memset(v, 0, sizeof *v); + if (in->op != USSE_MAD || in->pred || in->end || in->syncstart) + return 0; + /* Mask mode gates iterations, so a mask other than the first bit + * writes a register above the one named; the EFO the pass builds + * runs one iteration and writes the named one. */ + if (in->repeat > 1 || in->mask != 1) + return 0; + for (i = 0; i < 4; i++) + if (moe[i] & 0x100) + return 0; + v->d = in->dst; + if (is_const(&in->src[2], USSE_C_ZERO)) { + v->kind = EFO_MUL; + v->a = in->src[0]; + v->b = in->src[1]; + } else if (is_const(&in->src[1], USSE_C_ONE) && !in->src[1].neg) { + v->kind = EFO_ADD; + v->a = in->src[0]; + v->b = in->src[2]; + } else { + v->kind = EFO_MAD; + v->a = in->src[0]; + v->b = in->src[1]; + v->c = in->src[2]; + } + return 1; +} + +/* The registers one instruction reads or writes, as [first, last] per bank. + * Mirrors emit()'s own accounting so a range is never understated. */ +struct efo_range { uint8_t bank; uint8_t first, last; }; + +static unsigned efo_span(const struct usse_insn *in) +{ + if (in->repeat) + return in->repeat; + if (in->mask & 8) return 4; + if (in->mask & 4) return 3; + if (in->mask & 2) return 2; + if (in->mask & 1) return 1; + return usse_op_has_dst(in->op) ? 1 : 0; +} + +static unsigned efo_step(const uint16_t moe[4], unsigned slot, unsigned span) +{ + unsigned i, top = 0; + + if (!span) + return 0; + if (moe[slot] == MOE_INC0) + return 0; + if (!(moe[slot] & 0x100)) + return span - 1u; + for (i = 0; i < span && i < 4; i++) + if (chan(moe[slot] & 0xFFu, i) > top) + top = chan(moe[slot] & 0xFFu, i); + return top; +} + +static unsigned efo_reads(const struct usse_insn *in, const uint16_t moe[4], + struct efo_range *r) +{ + unsigned slots = usse_op_srcs(in->op), span = efo_span(in), i, n = 0; + + for (i = 0; i < 3; i++) { + unsigned reach; + + if (!((slots >> i) & 1)) + continue; + reach = efo_step(moe, i + 1u, span); + if (in->op == USSE_SMP && i == 0 && in->smp_dim > 1) + reach += in->smp_dim - 1u; + r[n].bank = in->src[i].bank; + r[n].first = in->src[i].num; + r[n].last = (uint8_t)(in->src[i].num + reach); + n++; + } + return n; +} + +static unsigned efo_writes(const struct usse_insn *in, const uint16_t moe[4], + struct efo_range *r) +{ + unsigned span = efo_span(in); + + if (!usse_op_has_dst(in->op) || !span) + return 0; + r[0].bank = in->dst.bank; + r[0].first = in->dst.num; + r[0].last = (uint8_t)(in->dst.num + efo_step(moe, 0, span)); + /* An EFO's implicit results are not in this list; the internal + * registers are tracked by efo_schedule() instead. */ + return 1; +} + +/* Whether this instruction certainly writes one named register. efo_writes() + * gives the range an operand may reach, which overstates a masked write - + * mask 0b0100 names three registers and writes the third. Ending the search + * for readers on the range would end it at a write that never happens and + * leave a later reader unaccounted for, so that search uses this instead. */ +static int efo_writes_reg(const struct usse_insn *in, const uint16_t moe[4], + unsigned bank, unsigned num) +{ + unsigned span = efo_span(in), i; + + if (!usse_op_has_dst(in->op) || !span || in->dst.bank != bank) + return 0; + for (i = 0; i < span; i++) { + unsigned at = in->dst.num; + + if (!in->repeat && !(in->mask & (1u << i))) + continue; + if (moe[0] == MOE_INC0) + ; + else if (moe[0] & 0x100) + at += chan(moe[0] & 0xFFu, i < 4 ? i : 3); + else + at += i; + if (at == num) + return 1; + } + return 0; +} + +static int efo_hits(const struct efo_range *r, unsigned n, unsigned bank, + unsigned lo, unsigned hi) +{ + unsigned i; + + for (i = 0; i < n; i++) + if (r[i].bank == bank && r[i].first <= hi && lo <= r[i].last) + return 1; + return 0; +} + +/* Does this instruction touch an internal register, either as an operand or + * as an EFO's implicit result? */ +static int efo_touches_ireg(const struct usse_insn *in) +{ + unsigned slots = usse_op_srcs(in->op), i; + + if (in->op == USSE_EFO) + return 1; + if (usse_op_has_dst(in->op) && in->dst.bank == USSE_INTERNAL) + return 1; + for (i = 0; i < 3; i++) + if (((slots >> i) & 1) && in->src[i].bank == USSE_INTERNAL) + return 1; + return 0; +} + +/* Instructions after which the part may switch execution thread, and so + * lose the internal registers: a WDF and anything carrying .syncs force one + * (usc2/finalise.c:65 IsDeschedule), and a branch or the end of the program + * ends the basic block, across which usc2 asserts the internal registers are + * never live (finalise.c:311, :770). A sample is counted with them: its + * dependent read is what the following WDF waits on. */ +static int efo_is_switch(const struct usse_insn *in) +{ + switch (in->op) { + case USSE_WDF: case USSE_BA: case USSE_BR: case USSE_EMIT: + case USSE_SMP: + return 1; + default: + return in->syncstart || in->end; + } +} + +/* Can this instruction read an internal register in the given source slot? + * Slot 0 reaches i0..i3 as the top four temporaries and slots 1 and 2 reach + * them through the extended bank; the pack, sample and test forms encode + * those slots differently, so they are left out. */ +static int efo_can_read_ireg(unsigned op) +{ + switch (op) { + case USSE_MAD: case USSE_ADM: case USSE_MSA: case USSE_SUBFLR: + case USSE_RCP: case USSE_RSQ: case USSE_LOG: case USSE_EXP: + case USSE_DP: case USSE_MIN: case USSE_MAX: + case USSE_MOV: case USSE_MOVC: + return 1; + default: + return 0; + } +} + +/* Can this instruction carry .nosched? A WDF has no such field, and usc2 + * refuses to put the flag on an instruction that itself forces a switch + * (finalise.c:648 ASSERT(!GetBit(..., INST_SYNCSTART))). */ +static int efo_can_nosched(const struct usse_insn *in) +{ + if (in->syncstart || in->op == USSE_WDF) + return 0; + return in->op == USSE_EFO || in->op == USSE_PCKU8F32 || + in->op == USSE_UNPCKF32U8 || in->op == USSE_PCKF16F32 || + in->op == USSE_UNPCKF32F16 || in->op == USSE_UNPCKF32U16 || + in->op == USSE_UNPCKF32S16 || in->op == USSE_LIMM || + usse_op_has_dst(in->op); +} + +/* Fill in the EFO that computes `p` into its own destination register and + * `q` into internal register *ireg. Returns 0 when no 535 wiring does it. */ +static int efo_build(const struct efo_val *p, const struct efo_val *q, + struct usse_insn *out, unsigned *ireg) +{ + struct usse_operand s0, s1, s2; + unsigned i; + + memset(out, 0, sizeof *out); + out->op = USSE_EFO; + out->repeat = 1; + out->dst = p->d; + + if (p->kind == EFO_MUL && q->kind == EFO_MUL) { + /* m0 = s0*s1, m1 = s0*s2 (MSRC 0): two products sharing one + * multiplicand, q in m0 and p in m1. */ + const struct usse_operand *sh = NULL, *po = NULL, *qo = NULL; + + if (efo_same(&p->a, &q->a)) { sh = &p->a; po = &p->b; qo = &q->b; } + else if (efo_same(&p->a, &q->b)) { sh = &p->a; po = &p->b; qo = &q->a; } + else if (efo_same(&p->b, &q->a)) { sh = &p->b; po = &p->a; qo = &q->b; } + else if (efo_same(&p->b, &q->b)) { sh = &p->b; po = &p->a; qo = &q->a; } + if (sh) { + s0 = *sh; s1 = *qo; s2 = *po; + out->efo_msrc = USSE_EFO_M_S0S1_S0S2; + out->efo_isrc = USSE_EFO_I_M0M1; + out->efo_dsrc = USSE_EFO_D_I1; + out->efo_wi0 = 1; + *ireg = 0; + goto place; + } + /* m0 = s1*s2, m1 = s0*s0 (MSRC 2): p as the free product and + * q as the square of one register. */ + if (efo_same(&q->a, &q->b)) { + s0 = q->a; s1 = p->a; s2 = p->b; + out->efo_msrc = USSE_EFO_M_S1S2_S0S0; + out->efo_isrc = USSE_EFO_I_M0M1; + out->efo_dsrc = USSE_EFO_D_I0; + out->efo_wi1 = 1; + *ireg = 1; + goto place; + } + if (efo_same(&p->a, &p->b)) { + s0 = p->a; s1 = q->a; s2 = q->b; + out->efo_msrc = USSE_EFO_M_S1S2_S0S0; + out->efo_isrc = USSE_EFO_I_M0M1; + out->efo_dsrc = USSE_EFO_D_I1; + out->efo_wi0 = 1; + *ireg = 0; + goto place; + } + return 0; + } + + if (p->kind == EFO_ADD && q->kind == EFO_ADD) { + /* a0 = s0+s1, a1 = s2+s0 (ASRC 3): two sums sharing one + * addend, p in a0 and q in a1. */ + const struct usse_operand *sh = NULL, *po = NULL, *qo = NULL; + + if (efo_same(&p->a, &q->a)) { sh = &p->a; po = &p->b; qo = &q->b; } + else if (efo_same(&p->a, &q->b)) { sh = &p->a; po = &p->b; qo = &q->a; } + else if (efo_same(&p->b, &q->a)) { sh = &p->b; po = &p->a; qo = &q->b; } + else if (efo_same(&p->b, &q->b)) { sh = &p->b; po = &p->a; qo = &q->a; } + if (!sh) + return 0; + s0 = *sh; s1 = *po; s2 = *qo; + out->efo_asrc = USSE_EFO_A_S0S1_S2S0; + out->efo_isrc = USSE_EFO_I_A0A1; + out->efo_dsrc = USSE_EFO_D_A0; + out->efo_wi1 = 1; + *ireg = 1; + goto place; + } + + if ((p->kind == EFO_ADD && q->kind == EFO_MUL) || + (p->kind == EFO_MUL && q->kind == EFO_ADD)) { + /* a0 = s0+s1 with m1 = s0*s2 (ASRC 3, MSRC 0, ISRC 3): a sum + * and a product sharing one operand. */ + const struct efo_val *ad = p->kind == EFO_ADD ? p : q; + const struct efo_val *mu = p->kind == EFO_ADD ? q : p; + const struct usse_operand *sh = NULL, *ao = NULL, *mo = NULL; + + if (efo_same(&ad->a, &mu->a)) { sh = &ad->a; ao = &ad->b; mo = &mu->b; } + else if (efo_same(&ad->a, &mu->b)) { sh = &ad->a; ao = &ad->b; mo = &mu->a; } + else if (efo_same(&ad->b, &mu->a)) { sh = &ad->b; ao = &ad->a; mo = &mu->b; } + else if (efo_same(&ad->b, &mu->b)) { sh = &ad->b; ao = &ad->a; mo = &mu->a; } + if (!sh) + return 0; + s0 = *sh; s1 = *ao; s2 = *mo; + out->efo_asrc = USSE_EFO_A_S0S1_S2S0; + out->efo_msrc = USSE_EFO_M_S0S1_S0S2; + out->efo_isrc = USSE_EFO_I_A0M1; + if (p->kind == EFO_ADD) { + out->efo_dsrc = USSE_EFO_D_I0; /* a0 */ + out->efo_wi1 = 1; + *ireg = 1; + } else { + out->efo_dsrc = USSE_EFO_D_I1; /* m1 */ + out->efo_wi0 = 1; + *ireg = 0; + } + goto place; + } + + if ((p->kind == EFO_MAD && q->kind == EFO_MUL) || + (p->kind == EFO_MUL && q->kind == EFO_MAD)) { + /* a0 = m0+s2 with m0 = s0*s1 and m1 = s0*s2 (ASRC 1, MSRC 0, + * ISRC 3): the vendor's sS0S1S0S2_M0S2. The product has to be + * the multiply-add's first multiplicand times its addend. */ + const struct efo_val *ma = p->kind == EFO_MAD ? p : q; + const struct efo_val *mu = p->kind == EFO_MAD ? q : p; + + if (efo_same(&mu->a, &ma->a) && efo_same(&mu->b, &ma->c)) { + s0 = ma->a; s1 = ma->b; s2 = ma->c; + } else if (efo_same(&mu->b, &ma->a) && efo_same(&mu->a, &ma->c)) { + s0 = ma->a; s1 = ma->b; s2 = ma->c; + } else { + return 0; + } + out->efo_asrc = USSE_EFO_A_M0S2_I1I0; + out->efo_msrc = USSE_EFO_M_S0S1_S0S2; + out->efo_isrc = USSE_EFO_I_A0M1; + if (p->kind == EFO_MAD) { + out->efo_dsrc = USSE_EFO_D_I0; /* a0 */ + out->efo_wi1 = 1; + *ireg = 1; + } else { + out->efo_dsrc = USSE_EFO_D_I1; /* m1 */ + out->efo_wi0 = 1; + *ireg = 0; + } + goto place; + } + return 0; + +place: + if (!efo_bank_ok(&s0, 0) || !efo_bank_ok(&s1, 1) || !efo_bank_ok(&s2, 2)) + return 0; + /* The whole point is to lose one instruction; an EFO whose + * destination is the internal bank would need one more to get the + * value out again. */ + if (out->dst.bank != USSE_TEMP && out->dst.bank != USSE_PA && + out->dst.bank != USSE_OUT) + return 0; + out->src[0] = s0; + out->src[1] = s1; + out->src[2] = s2; + out->skipinv = 1; + for (i = 0; i < 3; i++) + out->src[i].comp = 0; + return 1; +} + +/* A trial copy of the whole stream, so a fusion that turns out to be + * illegal is dropped rather than patched. */ +struct efo_trial { + struct usse_insn *ins; + uint16_t (*moe)[4]; + uint32_t *orig; /* where each instruction started */ + size_t n; + uint32_t *labelpc; + uint32_t nlab; +}; + +static int efo_label_in(const struct efo_trial *t, size_t lo, size_t hi) +{ + uint32_t i; + + for (i = 0; i < t->nlab; i++) + if (t->labelpc[i] != NOREG && t->labelpc[i] > lo && + t->labelpc[i] <= hi) + return 1; + return 0; +} + +/* Walk the internal registers' lifetimes and place the .nosched flags they + * need. Returns 0 when every lifetime can be kept, -1 when one cannot. */ +static int efo_schedule(struct efo_trial *t) +{ + size_t i, j, q; + unsigned r; + + /* Nothing else in this backend sets the flag, so recomputing it from + * scratch keeps it exactly where the lifetimes want it after a + * deletion has moved the parities. */ + for (i = 0; i < t->n; i++) + t->ins[i].nosched = 0; + + for (i = 0; i < t->n; i++) { + if (t->ins[i].op != USSE_EFO) + continue; + for (r = 0; r < 2; r++) { + size_t last = i; + + if (!(r ? t->ins[i].efo_wi1 : t->ins[i].efo_wi0)) + continue; + for (j = i + 1; j < t->n; j++) { + struct efo_range rd[3]; + unsigned nr = efo_reads(&t->ins[j], t->moe[j], rd); + + if (efo_hits(rd, nr, USSE_INTERNAL, r, r)) + last = j; + if (t->ins[j].op == USSE_EFO && + (r ? t->ins[j].efo_wi1 : t->ins[j].efo_wi0)) + break; + if (efo_writes_reg(&t->ins[j], t->moe[j], + USSE_INTERNAL, r)) + break; + } + if (last == i) + continue; /* written, never read */ + /* Nothing may deschedule, and no other path may join, + * while the register holds the value. */ + for (j = i; j < last; j++) + if (efo_is_switch(&t->ins[j])) + return -1; + if (efo_label_in(t, i, last)) + return -1; + /* A pair boundary sits after every odd slot. The flag + * disables descheduling at the end of the pair after + * the one carrying it, so it goes two pairs back. */ + for (q = i | 1u; q < last; q += 2) { + size_t p2; + + if (q < 3) + return -1; + p2 = q - 2; + if (!efo_can_nosched(&t->ins[p2])) + p2 = q - 3; + if (!efo_can_nosched(&t->ins[p2])) + return -1; + t->ins[p2].nosched = 1; + } + } + } + return 0; +} + +/* emit() splits two branches that would share a pair; after a deletion the + * parities have moved, so the result has to be checked again rather than + * padded - padding would cost back what the fusion saved. */ +static int efo_pairs_ok(const struct efo_trial *t) +{ + size_t k; + + for (k = 1; k < t->n; k += 2) + if (pair_ctrl(t->ins[k - 1].op) && pair_ctrl(t->ins[k].op)) + return -1; + return 0; +} + +static int efo_encodes(const struct efo_trial *t) +{ + size_t k; + uint64_t w; + + for (k = 0; k < t->n; k++) + if (usse_encode(&t->ins[k], &w, NULL) != 0) + return -1; + return 0; +} + +/* Every later instruction that reads one register of one bank, up to the + * point where it is written again. Returns the count, and the single reader + * in *reader when there is exactly one. */ +static unsigned efo_readers(const struct efo_trial *t, size_t from, + unsigned bank, unsigned num, size_t *reader) +{ + size_t j; + unsigned n = 0; + + for (j = from + 1; j < t->n; j++) { + struct efo_range rd[3]; + unsigned nr = efo_reads(&t->ins[j], t->moe[j], rd); + + if (efo_hits(rd, nr, bank, num, num)) { + if (!n) + *reader = j; + n++; + } + if (efo_writes_reg(&t->ins[j], t->moe[j], bank, num)) + break; + } + return n; +} + +/* Point the reader's sources at the internal register instead. Returns 0 + * when every slot that names the register can be redirected exactly. */ +static int efo_redirect(struct efo_trial *t, size_t r, unsigned bank, + unsigned num, unsigned ireg) +{ + struct usse_insn *in = &t->ins[r]; + unsigned slots = usse_op_srcs(in->op), span = efo_span(in), i, hit = 0; + + if (!efo_can_read_ireg(in->op)) + return -1; + for (i = 0; i < 3; i++) { + if (!((slots >> i) & 1)) + continue; + if (in->src[i].bank != bank) + continue; + if (in->src[i].num != num) { + /* The slot steps over a range that covers it; the + * internal register does not step, so it cannot + * stand in for one element of that range. */ + if (in->src[i].num <= num && + num <= in->src[i].num + efo_step(t->moe[r], i + 1u, span)) + return -1; + continue; + } + if (efo_step(t->moe[r], i + 1u, span) != 0) + return -1; + in->src[i].bank = USSE_INTERNAL; + in->src[i].num = (uint8_t)ireg; + hit++; + } + return hit ? 0 : -1; +} + +static void efo_delete(struct efo_trial *t, size_t at) +{ + uint32_t i; + + memmove(t->ins + at, t->ins + at + 1, + (t->n - at - 1) * sizeof *t->ins); + memmove(t->moe + at, t->moe + at + 1, + (t->n - at - 1) * sizeof *t->moe); + memmove(t->orig + at, t->orig + at + 1, + (t->n - at - 1) * sizeof *t->orig); + t->n--; + for (i = 0; i < t->nlab; i++) + if (t->labelpc[i] != NOREG && t->labelpc[i] > at) + t->labelpc[i]--; +} + +/* Try to fuse the instructions at a and b, with `keep` naming which of the + * two keeps its own destination register - the other one's result lives in + * an internal register and its single later reader is pointed at that. + * Returns 0 when the trial holds the fused stream, -1 when it is unchanged. */ +static enum efo_why efo_try(struct efo_trial *t, size_t a, size_t b, + size_t keep) +{ + struct efo_val va, vb; + const struct efo_val *p, *q; + struct usse_insn efo; + struct efo_range rd[3], wr[2]; + unsigned ireg, nr, nw, k; + size_t r = 0, j, from; + + if (!efo_classify(&t->ins[a], t->moe[a], &va) || + !efo_classify(&t->ins[b], t->moe[b], &vb)) + return EFO_NOT_ALU; + p = keep == a ? &va : &vb; + q = keep == a ? &vb : &va; + if (!efo_build(p, q, &efo, &ireg)) + return EFO_NO_SHAPE; + /* The result that goes to an internal register loses its own write, + * so it has to be a temporary read exactly once afterwards. */ + if (q->d.bank != USSE_TEMP || efo_same(&va.d, &vb.d)) + return EFO_NOT_DEAD; + nr = efo_reads(&t->ins[b], t->moe[b], rd); + if (efo_hits(rd, nr, va.d.bank, va.d.num, va.d.num)) + return EFO_DEPEND; /* b reads what a writes */ + nr = efo_reads(&t->ins[a], t->moe[a], rd); + if (efo_hits(rd, nr, vb.d.bank, vb.d.num, vb.d.num)) + return EFO_DEPEND; /* a reads what b writes */ + /* Both instructions now issue at a's slot, so nothing between may + * write what they read, deschedule, or touch the internal registers - + * and when b's destination is the one kept, nothing between may read + * or write it either, because it is written earlier than it was. */ + nr = efo_reads(&efo, t->moe[a], rd); + for (j = a + 1; j < b; j++) { + if (efo_is_switch(&t->ins[j]) || efo_touches_ireg(&t->ins[j])) + return EFO_DEPEND; + nw = efo_writes(&t->ins[j], t->moe[j], wr); + for (k = 0; k < nw; k++) + if (efo_hits(rd, nr, wr[k].bank, wr[k].first, wr[k].last)) + return EFO_DEPEND; + if (keep != b) + continue; + if (efo_hits(wr, nw, p->d.bank, p->d.num, p->d.num)) + return EFO_DEPEND; + nw = efo_reads(&t->ins[j], t->moe[j], wr); + if (efo_hits(wr, nw, p->d.bank, p->d.num, p->d.num)) + return EFO_DEPEND; + } + if (efo_label_in(t, a, b)) + return EFO_DEPEND; + from = keep == a ? b : a; + if (efo_readers(t, from, q->d.bank, q->d.num, &r) != 1 || r == b) + return EFO_NOT_DEAD; + if (efo_redirect(t, r, q->d.bank, q->d.num, ireg) != 0) + return EFO_READER; + t->ins[a] = efo; + efo_delete(t, b); + return EFO_OK; +} + +/* Behind SGX_EFO and off by default: nothing has run this on the part yet, + * and a wrong internal-register lifetime does not fault, it renders the + * wrong value or stalls the core. */ +static void efo_pass(struct cg *c) +{ + struct efo_trial cur, trial; + unsigned why[EFO_WHY_COUNT]; + size_t a, b; + uint32_t i; + + if (!c->efo || c->failed || c->n < 2 || !c->slist) + return; + memset(&cur, 0, sizeof cur); + memset(&trial, 0, sizeof trial); + memset(why, 0, sizeof why); + cur.n = trial.n = c->n; + cur.nlab = trial.nlab = c->s->next_label + 1u; + cur.ins = malloc(c->n * sizeof *cur.ins); + cur.moe = malloc(c->n * sizeof *cur.moe); + cur.orig = malloc(c->n * sizeof *cur.orig); + cur.labelpc = malloc(cur.nlab * sizeof *cur.labelpc); + trial.ins = malloc(c->n * sizeof *trial.ins); + trial.moe = malloc(c->n * sizeof *trial.moe); + trial.orig = malloc(c->n * sizeof *trial.orig); + trial.labelpc = malloc(trial.nlab * sizeof *trial.labelpc); + if (!cur.ins || !cur.moe || !cur.orig || !cur.labelpc || !trial.ins || + !trial.moe || !trial.orig || !trial.labelpc) + goto out; + memcpy(cur.ins, c->slist, c->n * sizeof *cur.ins); + memcpy(cur.moe, c->smoe, c->n * sizeof *cur.moe); + memcpy(cur.labelpc, c->labelpc, cur.nlab * sizeof *cur.labelpc); + for (i = 0; i < c->n; i++) + cur.orig[i] = i; + + for (a = 0; a + 1 < cur.n; a++) { + unsigned keep, took = 0; + + for (b = a + 1; !took && b < cur.n && b <= a + EFO_WINDOW; b++) + for (keep = 0; keep < 2; keep++) { + trial.n = cur.n; + memcpy(trial.ins, cur.ins, cur.n * sizeof *cur.ins); + memcpy(trial.moe, cur.moe, cur.n * sizeof *cur.moe); + memcpy(trial.orig, cur.orig, cur.n * sizeof *cur.orig); + memcpy(trial.labelpc, cur.labelpc, + cur.nlab * sizeof *cur.labelpc); + enum efo_why w = efo_try(&trial, a, b, keep ? b : a); + + if (w != EFO_OK) { + why[w]++; + continue; + } + if (efo_schedule(&trial) != 0) + w = EFO_SCHED; + else if (efo_pairs_ok(&trial) != 0) + w = EFO_PAIR; + else if (efo_encodes(&trial) != 0) + w = EFO_ENCODE; + if (w != EFO_OK) { + why[w]++; + continue; + } + why[EFO_OK]++; + memcpy(cur.ins, trial.ins, trial.n * sizeof *cur.ins); + memcpy(cur.moe, trial.moe, trial.n * sizeof *cur.moe); + memcpy(cur.orig, trial.orig, trial.n * sizeof *cur.orig); + memcpy(cur.labelpc, trial.labelpc, + cur.nlab * sizeof *cur.labelpc); + cur.n = trial.n; + took = 1; + break; + } + } + if (getenv("SGX_EFO_TRACE")) { + struct efo_val v; + unsigned cand = 0; + + for (a = 0; a < c->n; a++) + cand += efo_classify(&c->slist[a], c->smoe[a], &v); + fprintf(stderr, "sgx: efo: %zu instructions, %u fusable,", + c->n, cand); + for (i = 0; i < EFO_WHY_COUNT; i++) + fprintf(stderr, " %s %u", efo_why_name[i], why[i]); + fprintf(stderr, "\n"); + } + if (cur.n == c->n) + goto out; + + /* The branch fixups still name the addresses emission gave them. A + * branch is never one of the fused instructions, so each address is + * still in the stream; orig[] says where it moved to. */ + for (i = 0; i < c->nfix; i++) { + uint32_t k; + + for (k = 0; k < cur.n; k++) + if (cur.orig[k] == c->fix[i].at) { + c->fix[i].at = k; + break; + } + if (k == cur.n) { + fail(c, "efo: a branch fixup lost its instruction"); + goto out; + } + } + memcpy(c->slist, cur.ins, cur.n * sizeof *cur.ins); + memcpy(c->smoe, cur.moe, cur.n * sizeof *cur.moe); + memcpy(c->labelpc, cur.labelpc, cur.nlab * sizeof *cur.labelpc); + c->n = cur.n; + for (i = 0; i < c->n; i++) { + uint64_t w; + const char *e = NULL; + + if (usse_encode(&c->slist[i], &w, &e) != 0) { + fail(c, "efo: %s", e ? e : "encode failed"); + break; + } + if (c->out && i < c->cap) + c->out[i] = w; + if (c->listing && i < c->lcap) + c->listing[i] = c->slist[i]; + } +out: + free(cur.ins); free(cur.moe); free(cur.orig); free(cur.labelpc); + free(trial.ins); free(trial.moe); free(trial.orig); free(trial.labelpc); +} + +/* --- entry points ------------------------------------------------------ */ + +void uir_codegen_result_fini(struct uir_codegen_result *res) +{ + if (!res) + return; + free(res->uni_base); + free(res->uni_width); + free(res->smp_slot); + res->uni_base = NULL; + res->uni_width = NULL; + res->smp_slot = NULL; +} + +int usse_codegen_ex(const struct uir_shader *s, + const struct uir_codegen_opts *opts, + uint64_t *out, size_t out_cap, + struct uir_codegen_result *res, + struct usse_layout *lay, + struct usse_insn *listing, size_t listing_cap, + int allow_unlowered, + char *err, size_t errn) +{ + struct cg c; + struct uir_shader *split = NULL, *propped = NULL; + size_t i; + unsigned pass; + int rc; + + /* Zeroed here rather than where it is filled: res carries allocations + * and the caller releases them on every path, including the failures + * that return before the fill. */ + if (res) + memset(res, 0, sizeof *res); + if (!s) { + if (err && errn) snprintf(err, errn, "no shader"); + return -1; + } + memset(&c, 0, sizeof c); + c.s = s; + if (opts) c.opts = *opts; + c.out = out; + c.cap = out_cap; + c.listing = listing; + c.lcap = listing_cap; + c.lay = lay; + c.err = err; + c.errn = errn; + c.allow_unlowered = allow_unlowered; + c.fragment = s->stage == UIR_STAGE_FRAGMENT; + c.frag_out_packed = opts && opts->frag_out_packed; + c.frag_out_rgba = opts && opts->frag_out_rgba; + c.uses_kill = c.fragment && s->uses_kill; + c.frag_tex_preiterated = opts && opts->frag_tex_preiterated; + /* Zero is a real base, not "unset": the texel registers are handed out + * in issue order, so a program whose every set is a sampled coordinate + * - no iterated colour in front of them - has its first texel in pa0. + * Treating zero as unset put both texels one register too high and the + * second landed on one nothing wrote, which reads as one: every masked + * composite came out with the mask fully opaque. Negative means unset. */ + c.frag_tex_pa_base = (opts && opts->frag_tex_pa_base >= 0) ? + opts->frag_tex_pa_base : 1; + c.vtx_uniform_pa_base = opts ? opts->vtx_uniform_pa_base : 0; + c.vtx_no_pool = opts ? opts->vtx_no_literal_pool : 0; + c.frag_in_bases = opts && opts->frag_in_bases; + c.vtx_limm = opts ? opts->vtx_uniform_limm : 0; + /* A vertex program has no literal pool, but an immediate needs no + * attribute register either: it is written into the instruction + * stream in front of the instruction that reads it. That is what the + * uniform path already did, and it was reachable only when the driver + * put its uniforms there too - so a shader with a plain constant in it + * was refused outright. */ + c.vtx_imm = !c.fragment && (c.vtx_limm || + (opts && opts->vtx_no_literal_pool)); + c.vtx_out_slot = (opts && opts->vtx_out_slots) ? opts->vtx_out_slot : + NULL; + c.vtx_out_dw = opts && opts->vtx_out_dwords ? opts->vtx_out_dw : NULL; + c.moe_known = 1; /* reset state is increment 1 on every slot */ + /* Off until it has been measured on the part: an internal register + * lost to a thread switch renders a wrong value or stalls the core, + * and neither reports anything. */ + c.efo = opts && opts->efo ? opts->efo > 0 + : getenv("SGX_EFO") != NULL; + c.eff = getenv("SGX_EFF_TRACE") != NULL; + /* An A/B lever on one binary. Two builds compared against each other + * have already reported regressions this project did not have. */ + c.moe_old = getenv("SGX_MOE_OLD") != NULL; + if (err && errn) err[0] = 0; + + c.labelpc = calloc(s->next_label + 1, sizeof *c.labelpc); + /* Not zero: zero is instruction 0, so a branch to a label that was + * never emitted became a branch to the top of the program. */ + if (c.labelpc) + memset(c.labelpc, 0xff, + (size_t)(s->next_label + 1) * sizeof *c.labelpc); + if (!c.labelpc) { + if (err && errn) snprintf(err, errn, "out of memory"); + return -1; + } + /* Which labels a branch actually names. One that none does is not a + * join - control only falls into it - so the MOE state in force there + * is the state the instruction before it left, and need not be + * reprogrammed. Every shader this front end emits ends in such a + * label. */ + c.label_target = calloc(s->next_label + 1, 1); + if (!c.label_target) { + free(c.labelpc); + if (err && errn) snprintf(err, errn, "out of memory"); + return -1; + } + for (i = 0; i < s->ninsns; i++) { + const struct uir_insn *n = &s->insns[i]; + + if ((n->op == UIR_BR || n->op == UIR_BRC) && + n->branch_target <= s->next_label) + c.label_target[n->branch_target] = 1; + } + + /* Values as numbered fuse unrelated quantities under one index; split + * them before liveness sees them. Splitting is an optimisation, so a + * shader too large to analyse simply compiles as it stands. */ + /* Off by default. Splitting is sound on straight-line code and saves + * real registers - the X server's glyph shader falls from thirty-two + * to twenty-six - but on a branchy program it produces an allocation + * the hardware renders wrongly: glamor's composite, which nests an + * early return inside two ifs, returned its zero arm unconditionally + * and the suite's two glamor cases went from passing to failing. The + * webs and the intervals need re-deriving against real control flow + * before this can be trusted; SGX_SPLIT turns it on meanwhile. */ + /* Copy propagation runs before anything looks at liveness, so the + * allocator sees the shorter form. */ + if (!getenv("SGX_NO_COPYPROP")) { + struct uir_shader *cp = malloc(sizeof *cp); + struct uir_insn *ci = s->ninsns ? + malloc(s->ninsns * sizeof *ci) : NULL; + + if (cp && (ci || !s->ninsns)) { + *cp = *s; + if (s->ninsns) + memcpy(ci, s->insns, s->ninsns * sizeof *ci); + cp->insns = ci; + cp->insns_cap = s->ninsns; + { + int d = usse_ir_dot_scalarise(cp); + int k = usse_ir_copy_prop(cp); + + if (getenv("SGX_IR_TRACE")) + fprintf(stderr, "sgx: ir: %d dot(s) " + "scalarised, %d copy(s) " + "propagated\n", d, k); + } + propped = cp; + s = cp; + c.s = cp; + } else { + free(ci); + free(cp); + } + } + if (getenv("SGX_SPLIT") && !getenv("SGX_NO_SPLIT")) { + split = split_ranges(s); + if (split) { + char *saved = c.err; + uint32_t vsplit, vwhole; + + /* Live ranges are intervals, and two webs that never + * overlap in execution can still overlap as intervals - + * so the split form is not always the cheaper one. + * Allocate both and keep whichever is. The trials must + * not leave a diagnostic behind. */ + c.err = NULL; + c.s = split; + scan(&c); + alloc_regs(&c); + vsplit = c.failed ? NOREG : c.reg_values; + c.failed = 0; + free_scan(&c); + + c.s = s; + scan(&c); + alloc_regs(&c); + vwhole = c.failed ? NOREG : c.reg_values; + c.failed = 0; + free_scan(&c); + c.err = saved; + + if (vsplit <= vwhole) { + s = split; + c.s = split; + } else { + free(split->insns); + free(split); + split = NULL; + } + } + } + + scan(&c); + if (getenv("SGX_CHECK_LIVE")) + check_live(&c); + /* Allocate every scratch quad first, then once the code has said which + * of them it reaches, lay the registers out again for that many. The + * fixed four cost twenty registers of a thirty-one register file and + * put glamor's composite shader out of reach of the hardware. */ + c.scratch_regs = SCRATCH_QUADS * QUAD; + for (pass = 0; pass < 3 && !c.failed; pass++) { + if (pass) + reset_emit(&c); + c.relax_budget = (pass == 0); + alloc_regs(&c); + emit_kill_init(&c); + emit_in_unpack(&c); + for (i = 0; i < s->ninsns && !c.failed; i++) + lower_insn(&c, &s->insns[i], i); + c.relax_budget = 0; + if (c.failed) + break; + if (c.scratch_used < c.scratch_regs) + c.scratch_regs = c.scratch_used; + else if (pass) + break; /* held to the real count, nothing left to trim */ + } + + efo_pass(&c); + + /* Branch offsets are relative to the following instruction; the base is + * an assumption, see README "what is not verified". */ + for (i = 0; i < c.nfix && !c.failed; i++) { + struct usse_insn b; + uint64_t w; + const char *e = NULL; + long off; + + if (c.fix[i].target > s->next_label || + c.labelpc[c.fix[i].target] == 0xffffffffu) { + fail(&c, "branch to undefined label %u", + c.fix[i].target); + break; + } + /* The target is the branch's own address plus the offset, not + * one past it. Read off the SGX540 microkernel: over its + * sixteen relative branches, pc + off lands on a block head + * eleven times against two for pc + off + 1, and the clearest + * case has pc + off + 1 landing on an instruction that depends + * on the one the branch would have skipped. */ + off = (long)c.labelpc[c.fix[i].target] - (long)c.fix[i].at; + memset(&b, 0, sizeof b); + b.op = USSE_BR; + b.skipinv = 1; + b.pred = c.fix[i].pred; + b.offset = (int32_t)off; + if (usse_encode(&b, &w, &e) != 0) { + fail(&c, "branch: %s", e ? e : "encode failed"); + break; + } + if (out && c.fix[i].at < out_cap) + out[c.fix[i].at] = w; + if (listing && c.fix[i].at < listing_cap) + listing[c.fix[i].at] = b; + } + + /* A program the USSE would run off the end of is not a program. The + * driver checks this too, but every other consumer got a listing that + * looked like a success and locked the core when it ran. */ + if (!c.failed && (!c.n || !c.last_end)) { + fail(&c, c.n ? "last instruction lacks .end; it would run on" + : "empty program"); + } + + if (res) { + memset(res, 0, sizeof *res); + res->ntemps = c.ntemps; + res->src1_reg = c.fragout1; + res->has_src1 = c.has_fragout1 ? 1u : 0u; + res->nprimattr = c.npa; + /* Everything the driver must place in the sa bank: uniforms, + * the per-sampler state blocks, then the literal pool. */ + res->nsecattr = c.nsa; + /* Reported here and not in alloc_regs(): the layout is only + * final at this point, and printing it earlier gave a + * pool_base of zero for every program. pool_scalar says how + * many literals are one component, which is what packing the + * pool would return. */ + if (getenv("SGX_REG_BREAKDOWN")) { + uint32_t i, scal = 0; + + for (i = 0; i < c.npool; i++) + scal += c.pool_scalar[i] ? 1u : 0u; + fprintf(stderr, "sgx: sa: uniforms to %u, samplers %u, " + "pool %u (%u scalar) from %u, %u register(s) " + "of %u\n", c.uni_regs, c.nsamp, c.npool, + scal, c.pool_base, res->nsecattr, + (unsigned)UIR_SA_BANK); + for (i = 0; i < c.npool; i++) { + unsigned m = c.pool_used[i], b, rch = 0; + + for (b = 0; b < 4; b++) + if (m & (1u << b)) + rch = b + 1; + fprintf(stderr, "sgx: sa: pool %u used %x " + "reach %u at %u\n", i, m, rch, + c.pool[i].sa_base); + } + } + res->ninsns = (uint32_t)c.n; + { + uint32_t j, kept = 0; + + for (j = 0; j < c.npool; j++) + if (!c.pool_spill[j]) + res->pool[kept++] = c.pool[j]; + res->npool = kept; + } + res->pool_base = c.pool_base; + res->smp_base = c.smp_base; + res->nsamp = c.nsamp; + res->nsmp_slots = smp_slots_total(&c); + res->smp_slot = c.nsamp ? calloc(c.nsamp, 1) : NULL; + if (c.nsamp && !res->smp_slot) + fail(&c, "out of memory"); + for (i = 0; i < c.nsamp && res->smp_slot; i++) + res->smp_slot[i] = (unsigned char)smp_slot_of(&c, i); + res->nuniform = c.nuni; + res->uni_base = c.nuni ? calloc(c.nuni, 1) : NULL; + res->uni_width = c.nuni ? calloc(c.nuni, 1) : NULL; + if (c.nuni && (!res->uni_base || !res->uni_width)) { + fail(&c, "out of memory"); + } else { + for (i = 0; i < c.nuni; i++) { + res->uni_base[i] = c.uni_base[i]; + res->uni_width[i] = c.uni_reach[i] ? + c.uni_reach[i] : + (uint8_t)QUAD; + } + } + res->pool_dwords = QUAD * c.npool; + memcpy(res->limm, c.limm, c.nlimm * sizeof c.limm[0]); + res->nlimm = c.nlimm; + res->uses_kill = (uint32_t)c.uses_kill; + res->coord_f16 = c.coord_f16; + } + if (lay) { + for (i = 0; i < c.npool; i++) { + memcpy(lay->pool[i].v, c.pool[i].v, sizeof lay->pool[i].v); + lay->pool[i].sa_base = c.pool[i].sa_base; + } + lay->npool = c.npool; + lay->pool_base = c.pool_base; + } + rc = c.failed ? -1 : (int)c.n; + + free(c.uni_reach); + free(c.uni_umask); + free(c.uni_base); + free(c.vbase); + free(c.vfirst); + free(c.vlast); + free(c.vwidth); + free(c.vused); + free(c.scratch_at); + free(c.scratch_need); + free(c.scratch_w); + free(c.vorder); + if (split) { + free(split->insns); + free(split); + } + if (propped) { + free(propped->insns); + free(propped); + } + free(c.labelpc); + free(c.label_target); + free(c.fix); + free(c.slist); + free(c.smoe); + return rc; +} + +int uir_codegen(const struct uir_shader *s, const struct uir_codegen_opts *opts, + uint64_t *out, size_t out_cap, struct uir_codegen_result *res, + char *err, size_t errn) +{ + struct usse_layout lay; + int rc; + + memset(&lay, 0, sizeof lay); + rc = usse_codegen_ex(s, opts, out, out_cap, res, &lay, NULL, 0, 0, err, errn); + usse_layout_free(&lay); + return rc; +} + +void usse_layout_free(struct usse_layout *lay) +{ + if (!lay) + return; + free(lay->map); + lay->map = NULL; + lay->nmap = lay->map_cap = 0; +} diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/usse_ir.c mesa-26.2.2/src/gallium/drivers/sgx/usse_ir.c --- mesa-26.2.2.orig/src/gallium/drivers/sgx/usse_ir.c 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/usse_ir.c 2026-09-08 10:57:36.682292623 +0200 @@ -0,0 +1,1560 @@ +/* + * usse_ir.c - shared IR helpers, text form and validation for the SGX535 USSE compiler + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ + +#include +#include +#include +#include +#include + +#include "usse_ir.h" + +/* Every shader must come from uir_shader_new() or uir_parse(): uir_shader_free() + * recovers this prefix from the public pointer. */ +struct uir_priv { + char **strs; + size_t nstrs, strs_cap; + struct uir_shader pub; +}; + +static struct uir_priv *priv_of(struct uir_shader *s) +{ + return (struct uir_priv *)((char *)s - offsetof(struct uir_priv, pub)); +} + +/* --- tables ----------------------------------------------------------- */ + +static const char *const op_name[UIR_OP_COUNT] = { + "nop", "mov", "add", "sub", "mul", "mad", "min", "max", "frc", "floor", + "rcp", "rsq", "log2", "exp2", "sqrt", "dp3", "dp4", "cross", "nrm", + "slt", "sge", "seq", "sne", "cmp", "tex", "texproj", "texlod", "kill", + "emit", "br", "brc", "label", "ret", "ddx", "ddy", "face", + "texbias" +}; + +static const unsigned char op_nsrc[UIR_OP_COUNT] = { + 0, 1, 2, 2, 2, 3, 2, 2, 1, 1, + 1, 1, 1, 1, 1, 2, 2, 2, 1, + 2, 2, 2, 2, 3, 2, 2, 3, 1, + 0, 0, 1, 0, 0, + 1, 1, /* ddx, ddy */ + 0, /* face */ + 3 /* texbias */ +}; + +static const char *const type_name[] = { "f32", "i32", "i16", "i8", "i10" }; +static const char *const cls_name[] = { + "virt", "in", "uniform", "out", "imm", "texcoord", "sampler" +}; +static const char *const cls_pfx[] = { "v", "in", "u", "o", "", "tc", "s" }; + +static int op_has_dst(enum uir_op op) +{ + switch (op) { + case UIR_NOP: case UIR_KILL: case UIR_EMIT: + case UIR_BR: case UIR_BRC: case UIR_LABEL: case UIR_RET: + return 0; + default: + return 1; + } +} + +static int op_is_tex(enum uir_op op) +{ + return op == UIR_TEX || op == UIR_TEXPROJ || op == UIR_TEXLOD || + op == UIR_TEXBIAS; +} + +static const char *op_str(enum uir_op op) +{ + return ((int)op >= 0 && op < UIR_OP_COUNT) ? op_name[op] : "nop"; +} + +static const char *type_str(enum uir_type t) +{ + return ((int)t >= 0 && t <= UIR_I10) ? type_name[t] : "f32"; +} + +static int cls_ok(enum uir_regclass c) +{ + return (int)c >= 0 && c <= UIR_REG_SAMPLER; +} + +static void set_err(char *err, size_t errn, const char *fmt, ...) +{ + va_list ap; + + if (!err || !errn) + return; + va_start(ap, fmt); + vsnprintf(err, errn, fmt, ap); + va_end(ap); +} + +/* --- allocation ------------------------------------------------------- */ + +static char *str_dup_n(const char *s, size_t n) +{ + char *d = malloc(n + 1); + + if (!d) + return NULL; + memcpy(d, s, n); + d[n] = '\0'; + return d; +} + +static int priv_own(struct uir_shader *s, char *str) +{ + struct uir_priv *p = priv_of(s); + + if (p->nstrs == p->strs_cap) { + size_t nc = p->strs_cap ? p->strs_cap * 2 : 8; + char **n = realloc(p->strs, nc * sizeof *n); + + if (!n) + return -1; + p->strs = n; + p->strs_cap = nc; + } + p->strs[p->nstrs++] = str; + return 0; +} + +static char *priv_dup(struct uir_shader *s, const char *b, size_t n) +{ + char *d = str_dup_n(b, n); + + if (!d) + return NULL; + if (priv_own(s, d) < 0) { + free(d); + return NULL; + } + return d; +} + +struct uir_shader *uir_shader_new(enum uir_stage stage) +{ + struct uir_priv *p = calloc(1, sizeof *p); + + if (!p) + return NULL; + p->pub.stage = stage; + return &p->pub; +} + +void uir_shader_free(struct uir_shader *s) +{ + struct uir_priv *p; + size_t i; + + if (!s) + return; + p = priv_of(s); + for (i = 0; i < p->nstrs; i++) + free(p->strs[i]); + free(p->strs); + free(s->insns); + free(s->bindings); + free(p); +} + +static int insns_reserve(struct uir_shader *s, size_t need) +{ + size_t nc = s->insns_cap ? s->insns_cap : 16; + struct uir_insn *n; + + if (need <= s->insns_cap) + return 0; + while (nc < need) + nc *= 2; + n = realloc(s->insns, nc * sizeof *n); + if (!n) + return -1; + s->insns = n; + s->insns_cap = nc; + return 0; +} + +static int bindings_reserve(struct uir_shader *s, size_t need) +{ + size_t nc = s->bindings_cap ? s->bindings_cap : 8; + struct uir_binding *n; + + if (need <= s->bindings_cap) + return 0; + while (nc < need) + nc *= 2; + n = realloc(s->bindings, nc * sizeof *n); + if (!n) + return -1; + s->bindings = n; + s->bindings_cap = nc; + return 0; +} + +struct uir_insn *uir_emit(struct uir_shader *s, enum uir_op op) +{ + struct uir_insn *in; + int i; + + if (!s || insns_reserve(s, s->ninsns + 1) < 0) + return NULL; + in = &s->insns[s->ninsns++]; + memset(in, 0, sizeof *in); + in->op = op; + in->writemask = 0xf; + in->dst.swizzle = UIR_SWZ_XYZW; + for (i = 0; i < UIR_MAX_SRC; i++) + in->src[i].swizzle = UIR_SWZ_XYZW; + return in; +} + +uint32_t uir_alloc_virt(struct uir_shader *s) +{ + return s ? s->nvirt++ : 0; +} + +uint32_t uir_alloc_label(struct uir_shader *s) +{ + return s ? s->next_label++ : 0; +} + +int uir_add_binding(struct uir_shader *s, const struct uir_binding *b) +{ + if (!s || !b || bindings_reserve(s, s->nbindings + 1) < 0) + return -1; + s->bindings[s->nbindings++] = *b; + return 0; +} + +/* --- reference builders ----------------------------------------------- */ + +struct uir_ref uir_reg(enum uir_regclass cls, uint32_t index) +{ + struct uir_ref r; + + memset(&r, 0, sizeof r); + r.cls = cls; + r.index = index; + r.swizzle = UIR_SWZ_XYZW; + return r; +} + +struct uir_ref uir_imm4(float x, float y, float z, float w) +{ + struct uir_ref r = uir_reg(UIR_REG_IMM, 0); + + r.imm[0] = x; + r.imm[1] = y; + r.imm[2] = z; + r.imm[3] = w; + return r; +} + +struct uir_ref uir_imm1(float v) +{ + return uir_imm4(v, v, v, v); +} + +struct uir_ref uir_swz(struct uir_ref r, uint8_t swizzle) +{ + r.swizzle = swizzle; + return r; +} + +/* toggles, so uir_neg(uir_neg(r)) is r */ +struct uir_ref uir_neg(struct uir_ref r) +{ + r.negate = (uint8_t)!r.negate; + return r; +} + +/* --- text output ------------------------------------------------------ */ + +struct obuf { + char *buf; + size_t cap, len; +}; + +static void ob_putc(struct obuf *o, char c) +{ + if (o->buf && o->len + 1 < o->cap) + o->buf[o->len] = c; + o->len++; +} + +static void ob_puts(struct obuf *o, const char *s) +{ + for (; *s; s++) + ob_putc(o, *s); +} + +static void ob_printf(struct obuf *o, const char *fmt, ...) +{ + char tmp[128]; + va_list ap; + int n; + + va_start(ap, fmt); + n = vsnprintf(tmp, sizeof tmp, fmt, ap); + va_end(ap); + if (n < 0) + return; + if ((size_t)n >= sizeof tmp) { + o->len += (size_t)n; + return; + } + ob_puts(o, tmp); +} + +static void ob_finish(struct obuf *o) +{ + if (o->buf && o->cap) + o->buf[o->len < o->cap ? o->len : o->cap - 1] = '\0'; +} + +static void ob_quoted(struct obuf *o, const char *s) +{ + ob_putc(o, '"'); + for (; *s; s++) { + if (*s == '\\' || *s == '"') + ob_putc(o, '\\'); + ob_putc(o, *s); + } + ob_putc(o, '"'); +} + +static void ob_comment(struct obuf *o, const char *s) +{ + ob_puts(o, " ; "); + for (; *s; s++) + ob_putc(o, *s == '\n' ? ' ' : *s); +} + +static void print_ref(struct obuf *o, const struct uir_ref *r) +{ + int i; + + if (r->negate) + ob_putc(o, '-'); + if (r->absolute) + ob_putc(o, '|'); + if (r->cls == UIR_REG_IMM) { + ob_puts(o, "#("); + for (i = 0; i < 4; i++) { + if (i) + ob_putc(o, ','); + ob_printf(o, "%.9g", r->imm[i]); + } + ob_putc(o, ')'); + } else { + ob_printf(o, "%s%lu", cls_ok(r->cls) ? cls_pfx[r->cls] : "v", + (unsigned long)r->index); + } + if (r->absolute) + ob_putc(o, '|'); + if (r->swizzle != UIR_SWZ_XYZW) { + ob_putc(o, '.'); + for (i = 0; i < 4; i++) + ob_putc(o, "xyzw"[(r->swizzle >> (2 * i)) & 3]); + } +} + +static void print_insn(struct obuf *o, const struct uir_insn *in) +{ + unsigned i, n; + int hd; + + if (in->op == UIR_LABEL) { + ob_printf(o, "label %lu:", (unsigned long)in->label_id); + if (in->comment) + ob_comment(o, in->comment); + ob_putc(o, '\n'); + return; + } + + hd = op_has_dst(in->op); + ob_printf(o, " %s.%s", op_str(in->op), type_str(in->type)); + if (in->saturate) + ob_puts(o, ".sat"); + if (hd && in->writemask != 0xf) { + ob_putc(o, '.'); + if (!in->writemask) { + ob_puts(o, "none"); + } else { + for (i = 0; i < 4; i++) + if (in->writemask & (1u << i)) + ob_putc(o, "xyzw"[i]); + } + } + if (hd) { + ob_putc(o, ' '); + print_ref(o, &in->dst); + } + n = in->nsrc < UIR_MAX_SRC ? in->nsrc : UIR_MAX_SRC; + for (i = 0; i < n; i++) { + ob_puts(o, (i || hd) ? ", " : " "); + print_ref(o, &in->src[i]); + } + if (in->op == UIR_BR || in->op == UIR_BRC) { + ob_puts(o, (hd || n) ? ", " : " "); + ob_printf(o, "@%lu", (unsigned long)in->branch_target); + } + if (in->comment) + ob_comment(o, in->comment); + ob_putc(o, '\n'); +} + +int uir_print(const struct uir_shader *s, char *buf, size_t n) +{ + struct obuf o; + size_t i; + + o.buf = buf; + o.cap = n; + o.len = 0; + if (!s) { + ob_finish(&o); + return 0; + } + + ob_puts(&o, "# usse-ir 1\n"); + ob_printf(&o, "stage %s\n", + s->stage == UIR_STAGE_FRAGMENT ? "fragment" : "vertex"); + if (s->uses_kill) + ob_puts(&o, "kill 1\n"); + ob_printf(&o, "virt %lu\n", (unsigned long)s->nvirt); + ob_printf(&o, "labels %lu\n", (unsigned long)s->next_label); + + for (i = 0; i < s->nbindings; i++) { + const struct uir_binding *b = &s->bindings[i]; + + ob_printf(&o, "bind %s %lu slot %lu comp %u", + cls_ok(b->cls) ? cls_name[b->cls] : "virt", + (unsigned long)b->index, (unsigned long)b->slot, + (unsigned)b->components); + if (b->name) { + ob_puts(&o, " name "); + ob_quoted(&o, b->name); + } + ob_putc(&o, '\n'); + } + + for (i = 0; i < s->ninsns; i++) + print_insn(&o, &s->insns[i]); + + ob_finish(&o); + return (int)o.len; +} + +/* --- text input ------------------------------------------------------- */ + +static int is_ws(char c) +{ + return c == ' ' || c == '\t' || c == '\r'; +} + +static const char *skip_ws(const char *p, const char *e) +{ + while (p < e && is_ws(*p)) + p++; + return p; +} + +static const char *next_tok(const char *p, const char *e, + const char **tb, size_t *tl) +{ + p = skip_ws(p, e); + *tb = p; + while (p < e && !is_ws(*p)) + p++; + *tl = (size_t)(p - *tb); + return p; +} + +static int tok_is(const char *b, size_t n, const char *s) +{ + return strlen(s) == n && memcmp(b, s, n) == 0; +} + +static int tok_u32(const char *b, size_t n, uint32_t *out) +{ + uint32_t v = 0; + size_t i; + + if (!n) + return -1; + for (i = 0; i < n; i++) { + if (b[i] < '0' || b[i] > '9') + return -1; + v = v * 10 + (uint32_t)(b[i] - '0'); + } + *out = v; + return 0; +} + +static const char *scan_u32(const char *p, const char *e, uint32_t *out) +{ + uint32_t v = 0; + int n = 0; + + while (p < e && *p >= '0' && *p <= '9') { + v = v * 10 + (uint32_t)(*p - '0'); + p++; + n++; + } + if (!n) + return NULL; + *out = v; + return p; +} + +static int swz_index(char c) +{ + switch (c) { + case 'x': return 0; + case 'y': return 1; + case 'z': return 2; + case 'w': return 3; + default: return -1; + } +} + +static int mask_parse(const char *b, size_t n, uint8_t *out) +{ + uint8_t m = 0; + size_t i; + + if (!n || n > 4) + return -1; + for (i = 0; i < n; i++) { + int c = swz_index(b[i]); + + if (c < 0) + return -1; + m |= (uint8_t)(1 << c); + } + *out = m; + return 0; +} + +static int op_lookup(const char *b, size_t n) +{ + int i; + + for (i = 0; i < UIR_OP_COUNT; i++) + if (tok_is(b, n, op_name[i])) + return i; + return -1; +} + +static int type_lookup(const char *b, size_t n) +{ + int i; + + for (i = 0; i <= UIR_I10; i++) + if (tok_is(b, n, type_name[i])) + return i; + return -1; +} + +static int cls_lookup(const char *b, size_t n) +{ + int i; + + for (i = 0; i <= UIR_REG_SAMPLER; i++) + if (tok_is(b, n, cls_name[i])) + return i; + return -1; +} + +static int perr(char *err, size_t errn, int line, const char *msg) +{ + set_err(err, errn, "line %d: %s", line, msg); + return -1; +} + +static const char *line_end(const char *p) +{ + while (*p && *p != '\n') + p++; + return p; +} + +static const char *parse_ref(const char *p, const char *e, struct uir_ref *r) +{ + static const struct { + const char *pfx; + enum uir_regclass cls; + } pfx_tab[] = { + { "tc", UIR_REG_TEXCOORD }, { "in", UIR_REG_IN }, + { "v", UIR_REG_VIRT }, { "u", UIR_REG_UNIFORM }, + { "o", UIR_REG_OUT }, { "s", UIR_REG_SAMPLER } + }; + size_t k; + int i; + + *r = uir_reg(UIR_REG_VIRT, 0); + if (p < e && *p == '-') { + r->negate = 1; + p = skip_ws(p + 1, e); + } + if (p < e && *p == '|') { + r->absolute = 1; + p = skip_ws(p + 1, e); + } + + if (p < e && *p == '#') { + if (++p >= e || *p++ != '(') + return NULL; + r->cls = UIR_REG_IMM; + for (i = 0; i < 4; i++) { + char *end; + + p = skip_ws(p, e); + r->imm[i] = strtof(p, &end); + if (end == p || end > e) + return NULL; + p = skip_ws(end, e); + if (i < 3) { + if (p >= e || *p != ',') + return NULL; + p = skip_ws(p + 1, e); + } + } + if (p >= e || *p != ')') + return NULL; + p++; + } else { + for (k = 0; k < sizeof pfx_tab / sizeof pfx_tab[0]; k++) { + size_t l = strlen(pfx_tab[k].pfx); + + if ((size_t)(e - p) < l || memcmp(p, pfx_tab[k].pfx, l)) + continue; + r->cls = pfx_tab[k].cls; + p = scan_u32(p + l, e, &r->index); + break; + } + if (k == sizeof pfx_tab / sizeof pfx_tab[0] || !p) + return NULL; + } + + if (r->absolute) { + if (p >= e || *p != '|') + return NULL; + p++; + } + if (p < e && *p == '.') { + uint8_t sw = 0; + int n = 0, last = 0; + + p++; + while (p < e && n < 4) { + int c = swz_index(*p); + + if (c < 0) + break; + last = c; + sw |= (uint8_t)(c << (2 * n)); + n++; + p++; + } + if (!n) + return NULL; + /* a short swizzle replicates its last channel */ + for (; n < 4; n++) + sw |= (uint8_t)(last << (2 * n)); + r->swizzle = sw; + } + return p; +} + +static int parse_binding(struct uir_shader *s, const char *p, const char *e, + int line, char *err, size_t errn) +{ + struct uir_binding b; + const char *tb; + size_t tl; + uint32_t v; + int c; + + memset(&b, 0, sizeof b); + p = next_tok(p, e, &tb, &tl); + c = cls_lookup(tb, tl); + if (c < 0) + return perr(err, errn, line, "unknown register class in bind"); + b.cls = (enum uir_regclass)c; + + p = next_tok(p, e, &tb, &tl); + if (tok_u32(tb, tl, &b.index) < 0) + return perr(err, errn, line, "bad bind index"); + p = next_tok(p, e, &tb, &tl); + if (!tok_is(tb, tl, "slot")) + return perr(err, errn, line, "expected slot in bind"); + p = next_tok(p, e, &tb, &tl); + if (tok_u32(tb, tl, &b.slot) < 0) + return perr(err, errn, line, "bad bind slot"); + p = next_tok(p, e, &tb, &tl); + if (!tok_is(tb, tl, "comp")) + return perr(err, errn, line, "expected comp in bind"); + p = next_tok(p, e, &tb, &tl); + if (tok_u32(tb, tl, &v) < 0) + return perr(err, errn, line, "bad bind comp"); + b.components = (uint8_t)v; + + p = next_tok(p, e, &tb, &tl); + if (tl) { + char *name; + size_t n = 0; + + if (!tok_is(tb, tl, "name")) + return perr(err, errn, line, "expected name in bind"); + p = skip_ws(tb + tl, e); + if (p >= e || *p != '"') + return perr(err, errn, line, "expected quoted name"); + p++; + name = malloc((size_t)(e - p) + 1); + if (!name) + return perr(err, errn, line, "out of memory"); + while (p < e && *p != '"') { + if (*p == '\\' && p + 1 < e) + p++; + name[n++] = *p++; + } + if (p >= e) { + free(name); + return perr(err, errn, line, "unterminated name"); + } + name[n] = '\0'; + if (priv_own(s, name) < 0) { + free(name); + return perr(err, errn, line, "out of memory"); + } + b.name = name; + } + + if (uir_add_binding(s, &b) < 0) + return perr(err, errn, line, "out of memory"); + return 0; +} + +static int parse_insn(struct uir_shader *s, const char *tb, size_t tl, + const char *p, const char *e, int line, + char *err, size_t errn) +{ + struct uir_ref refs[UIR_MAX_SRC + 1]; + const char *semi = NULL, *q, *opend; + struct uir_insn *in; + uint32_t target = 0; + size_t i, ml, nrefs = 0; + int op, have_type = 0, have_target = 0; + + for (q = p; q < e; q++) + if (*q == ';') { + semi = q; + break; + } + opend = semi ? semi : e; + + if (tok_is(tb, tl, "label")) { + uint32_t id; + const char *lb; + size_t ll; + + next_tok(p, opend, &lb, &ll); + if (ll < 2 || lb[ll - 1] != ':' || tok_u32(lb, ll - 1, &id) < 0) + return perr(err, errn, line, "bad label"); + in = uir_emit(s, UIR_LABEL); + if (!in) + return perr(err, errn, line, "out of memory"); + in->label_id = id; + goto comment; + } + + ml = 0; + while (ml < tl && tb[ml] != '.') + ml++; + op = op_lookup(tb, ml); + if (op < 0) + return perr(err, errn, line, "unknown mnemonic"); + + in = uir_emit(s, (enum uir_op)op); + if (!in) + return perr(err, errn, line, "out of memory"); + + i = ml; + while (i < tl) { + const char *pb; + size_t pl = 0; + uint8_t m; + int t; + + i++; /* skip the '.' */ + pb = tb + i; + while (i < tl && tb[i] != '.') { + i++; + pl++; + } + if (!pl) + return perr(err, errn, line, "empty modifier"); + t = type_lookup(pb, pl); + if (t >= 0) { + in->type = (enum uir_type)t; + have_type = 1; + } else if (tok_is(pb, pl, "sat")) { + in->saturate = 1; + } else if (tok_is(pb, pl, "none")) { + in->writemask = 0; + } else if (mask_parse(pb, pl, &m) == 0) { + in->writemask = m; + } else { + return perr(err, errn, line, "unknown modifier"); + } + } + if (!have_type) + return perr(err, errn, line, "missing type suffix"); + + p = skip_ws(p, opend); + while (p < opend) { + if (*p == '@') { + p = scan_u32(p + 1, opend, &target); + if (!p) + return perr(err, errn, line, "bad branch target"); + have_target = 1; + } else { + if (nrefs >= UIR_MAX_SRC + 1) + return perr(err, errn, line, "too many operands"); + p = parse_ref(p, opend, &refs[nrefs++]); + if (!p) + return perr(err, errn, line, "bad operand"); + } + p = skip_ws(p, opend); + if (p < opend) { + if (*p != ',') + return perr(err, errn, line, "expected ','"); + p = skip_ws(p + 1, opend); + } + } + + if (op_has_dst(in->op)) { + if (!nrefs) + return perr(err, errn, line, "missing destination"); + in->dst = refs[0]; + for (i = 1; i < nrefs; i++) + in->src[i - 1] = refs[i]; + in->nsrc = (unsigned)(nrefs - 1); + } else { + for (i = 0; i < nrefs; i++) + in->src[i] = refs[i]; + in->nsrc = (unsigned)nrefs; + } + if (have_target) + in->branch_target = target; + +comment: + if (semi) { + const char *cb = semi + 1; + char *c; + + if (cb < e && *cb == ' ') + cb++; + c = priv_dup(s, cb, (size_t)(e - cb)); + if (!c) + return perr(err, errn, line, "out of memory"); + in->comment = c; + } + return 0; +} + +static int parse_line(struct uir_shader *s, const char *p, const char *e, + int line, char *err, size_t errn) +{ + const char *tb; + size_t tl; + uint32_t v; + + p = skip_ws(p, e); + if (p >= e || *p == '#') + return 0; + p = next_tok(p, e, &tb, &tl); + + if (tok_is(tb, tl, "stage")) { + next_tok(p, e, &tb, &tl); + if (tok_is(tb, tl, "vertex")) + s->stage = UIR_STAGE_VERTEX; + else if (tok_is(tb, tl, "fragment")) + s->stage = UIR_STAGE_FRAGMENT; + else + return perr(err, errn, line, "unknown stage"); + return 0; + } + if (tok_is(tb, tl, "kill")) { + next_tok(p, e, &tb, &tl); + if (tok_u32(tb, tl, &v) < 0) + return perr(err, errn, line, "bad kill value"); + s->uses_kill = (uint8_t)v; + return 0; + } + if (tok_is(tb, tl, "virt")) { + next_tok(p, e, &tb, &tl); + if (tok_u32(tb, tl, &s->nvirt) < 0) + return perr(err, errn, line, "bad virt count"); + return 0; + } + if (tok_is(tb, tl, "labels")) { + next_tok(p, e, &tb, &tl); + if (tok_u32(tb, tl, &s->next_label) < 0) + return perr(err, errn, line, "bad label count"); + return 0; + } + if (tok_is(tb, tl, "bind")) + return parse_binding(s, p, e, line, err, errn); + + return parse_insn(s, tb, tl, p, e, line, err, errn); +} + +struct uir_shader *uir_parse(const char *text, char *err, size_t errn) +{ + struct uir_shader *s; + const char *p; + int line = 0; + + if (!text) { + set_err(err, errn, "line 0: no input"); + return NULL; + } + s = uir_shader_new(UIR_STAGE_VERTEX); + if (!s) { + set_err(err, errn, "line 0: out of memory"); + return NULL; + } + + for (p = text; *p; ) { + const char *e = line_end(p); + + line++; + if (parse_line(s, p, e, line, err, errn) < 0) { + uir_shader_free(s); + return NULL; + } + p = (*e == '\n') ? e + 1 : e; + } + return s; +} + +/* --- validation ------------------------------------------------------- */ + +static int verr(char *err, size_t errn, size_t i, const char *msg) +{ + set_err(err, errn, "insn %lu: %s", (unsigned long)i, msg); + return -1; +} + +int uir_validate(const struct uir_shader *s, char *err, size_t errn) +{ + unsigned char *written; + int has_label = 0; + size_t i, j; + + if (!s) { + set_err(err, errn, "null shader"); + return -1; + } + + for (i = 0; i < s->ninsns; i++) { + if (s->insns[i].op != UIR_LABEL) + continue; + has_label = 1; + for (j = 0; j < i; j++) + if (s->insns[j].op == UIR_LABEL && + s->insns[j].label_id == s->insns[i].label_id) + return verr(err, errn, i, "duplicate label id"); + } + + for (i = 0; i < s->ninsns; i++) { + const struct uir_insn *in = &s->insns[i]; + int tex; + + if ((int)in->op < 0 || in->op >= UIR_OP_COUNT) + return verr(err, errn, i, "invalid opcode"); + if ((int)in->type < 0 || in->type > UIR_I10) + return verr(err, errn, i, "invalid type"); + if (in->nsrc > UIR_MAX_SRC) + return verr(err, errn, i, "too many sources"); + if (in->nsrc != (unsigned)op_nsrc[in->op]) + return verr(err, errn, i, "wrong source count"); + if (in->writemask > 0xf) + return verr(err, errn, i, "invalid writemask"); + + if (in->dst.cls == UIR_REG_IMM || in->dst.cls == UIR_REG_SAMPLER) + return verr(err, errn, i, "invalid destination class"); + if (op_has_dst(in->op)) { + if (in->dst.cls != UIR_REG_VIRT && in->dst.cls != UIR_REG_OUT) + return verr(err, errn, i, "invalid destination class"); + if (in->dst.cls == UIR_REG_VIRT && in->dst.index >= s->nvirt) + return verr(err, errn, i, "virtual register out of range"); + } + + tex = op_is_tex(in->op); + if (tex && (in->nsrc < 2 || in->src[1].cls != UIR_REG_SAMPLER)) + return verr(err, errn, i, "texture op needs a sampler source"); + + for (j = 0; j < in->nsrc; j++) { + const struct uir_ref *r = &in->src[j]; + + if (!cls_ok(r->cls)) + return verr(err, errn, i, "invalid source class"); + if (r->cls == UIR_REG_SAMPLER && !(tex && j == 1)) + return verr(err, errn, i, "misplaced sampler operand"); + if (r->cls == UIR_REG_VIRT && r->index >= s->nvirt) + return verr(err, errn, i, "virtual register out of range"); + } + + if (in->op == UIR_BR || in->op == UIR_BRC) { + int found = 0; + + for (j = 0; j < s->ninsns && !found; j++) + found = s->insns[j].op == UIR_LABEL && + s->insns[j].label_id == in->branch_target; + if (!found) + return verr(err, errn, i, "branch to missing label"); + } + } + + /* labels make a linear scan meaningless, so only then skip this check */ + if (has_label || !s->nvirt) + return 0; + written = calloc(s->nvirt, 1); + if (!written) { + set_err(err, errn, "out of memory"); + return -1; + } + for (i = 0; i < s->ninsns; i++) { + const struct uir_insn *in = &s->insns[i]; + + for (j = 0; j < in->nsrc; j++) + if (in->src[j].cls == UIR_REG_VIRT && + !written[in->src[j].index]) { + free(written); + return verr(err, errn, i, "undefined register read"); + } + if (op_has_dst(in->op) && in->dst.cls == UIR_REG_VIRT) + written[in->dst.index] = 1; + } + free(written); + return 0; +} + +/* --- self test -------------------------------------------------------- */ + +#ifdef UIR_IR_SELFTEST + +static int nfail; + +static char *print_alloc(const struct uir_shader *s, const char *tag) +{ + char small[8]; + char *b; + int n, m, k; + + n = uir_print(s, NULL, 0); + k = uir_print(s, small, sizeof small); + if (k != n || strlen(small) >= sizeof small) { + printf(" %s: truncated print returned %d, want %d\n", tag, k, n); + nfail++; + } + b = malloc((size_t)n + 1); + if (!b) { + printf(" %s: out of memory\n", tag); + nfail++; + return NULL; + } + m = uir_print(s, b, (size_t)n + 1); + if (m != n || strlen(b) != (size_t)n) { + printf(" %s: sizing mismatch %d vs %d\n", tag, n, m); + nfail++; + } + return b; +} + +static int str_eq(const char *a, const char *b) +{ + if (!a || !b) + return a == b; + return strcmp(a, b) == 0; +} + +static int ref_eq(const struct uir_ref *a, const struct uir_ref *b) +{ + return a->cls == b->cls && a->index == b->index && + a->swizzle == b->swizzle && a->negate == b->negate && + a->absolute == b->absolute && + memcmp(a->imm, b->imm, sizeof a->imm) == 0; +} + +static int cmp_shader(const struct uir_shader *a, const struct uir_shader *b) +{ + size_t i, j; + int bad = 0; + + if (a->stage != b->stage) + printf(" stage differs\n"), bad++; + if (a->ninsns != b->ninsns) { + printf(" ninsns %lu vs %lu\n", (unsigned long)a->ninsns, + (unsigned long)b->ninsns); + return 1; + } + if (a->nvirt != b->nvirt) + printf(" nvirt differs\n"), bad++; + if (a->next_label != b->next_label) + printf(" next_label differs\n"), bad++; + if (a->uses_kill != b->uses_kill) + printf(" uses_kill differs\n"), bad++; + if (a->nbindings != b->nbindings) { + printf(" nbindings %lu vs %lu\n", (unsigned long)a->nbindings, + (unsigned long)b->nbindings); + return 1; + } + + for (i = 0; i < a->nbindings; i++) { + const struct uir_binding *x = &a->bindings[i], *y = &b->bindings[i]; + + if (x->cls != y->cls || x->index != y->index || x->slot != y->slot || + x->components != y->components || !str_eq(x->name, y->name)) + printf(" binding %lu differs\n", (unsigned long)i), bad++; + } + + for (i = 0; i < a->ninsns; i++) { + const struct uir_insn *x = &a->insns[i], *y = &b->insns[i]; + + if (x->op != y->op || x->type != y->type || x->nsrc != y->nsrc || + x->writemask != y->writemask || x->saturate != y->saturate || + x->branch_target != y->branch_target || + x->label_id != y->label_id || !str_eq(x->comment, y->comment)) { + printf(" insn %lu header differs\n", (unsigned long)i); + bad++; + continue; + } + if (!ref_eq(&x->dst, &y->dst)) + printf(" insn %lu dst differs\n", (unsigned long)i), bad++; + for (j = 0; j < UIR_MAX_SRC; j++) + if (!ref_eq(&x->src[j], &y->src[j])) + printf(" insn %lu src%lu differs\n", + (unsigned long)i, (unsigned long)j), bad++; + } + return bad; +} + +static int roundtrip(struct uir_shader *s, const char *tag) +{ + struct uir_shader *r; + char err[128] = ""; + char *t1, *t2; + int bad = 0; + + t1 = print_alloc(s, tag); + if (!t1) + return 1; + r = uir_parse(t1, err, sizeof err); + if (!r) { + printf(" %s: parse failed: %s\n", tag, err); + free(t1); + nfail++; + return 1; + } + t2 = print_alloc(r, tag); + if (t2 && strcmp(t1, t2)) { + size_t i; + + for (i = 0; t1[i] && t2[i] && t1[i] == t2[i]; i++) + ; + printf(" %s: text differs at offset %lu\n", tag, (unsigned long)i); + bad++; + } + bad += cmp_shader(s, r); + free(t1); + free(t2); + uir_shader_free(r); + if (bad) + nfail++; + printf("%s: %s\n", bad ? "FAIL" : "PASS", tag); + return bad; +} + +static struct uir_shader *build_vertex(void) +{ + struct uir_shader *s = uir_shader_new(UIR_STAGE_VERTEX); + struct uir_binding b; + struct uir_insn *in; + uint32_t v0, v1, v2; + + memset(&b, 0, sizeof b); + b.cls = UIR_REG_IN; b.index = 0; b.slot = 0; b.components = 4; + b.name = "vPosition"; + uir_add_binding(s, &b); + b.cls = UIR_REG_IN; b.index = 1; b.slot = 1; b.components = 3; + b.name = "a \"quoted\\odd\" name"; + uir_add_binding(s, &b); + b.cls = UIR_REG_UNIFORM; b.index = 2; b.slot = 16; b.components = 4; + b.name = NULL; + uir_add_binding(s, &b); + b.cls = UIR_REG_OUT; b.index = 0; b.slot = 0; b.components = 4; + b.name = "gl_Position"; + uir_add_binding(s, &b); + b.cls = UIR_REG_TEXCOORD; b.index = 3; b.slot = 3; b.components = 2; + b.name = NULL; + uir_add_binding(s, &b); + + v0 = uir_alloc_virt(s); + v1 = uir_alloc_virt(s); + v2 = uir_alloc_virt(s); + + in = uir_emit(s, UIR_MOV); + in->dst = uir_reg(UIR_REG_VIRT, v0); + in->src[0] = uir_reg(UIR_REG_IN, 0); + in->nsrc = 1; + in->comment = "load position"; + + in = uir_emit(s, UIR_MAD); + in->dst = uir_reg(UIR_REG_VIRT, v1); + in->writemask = 0x7; + in->src[0] = uir_swz(uir_reg(UIR_REG_IN, 1), UIR_SWZ_XXXX); + in->src[1] = uir_neg(uir_reg(UIR_REG_UNIFORM, 2)); + in->src[2] = uir_reg(UIR_REG_VIRT, v0); + in->src[2].absolute = 1; + in->nsrc = 3; + + in = uir_emit(s, UIR_CROSS); + in->dst = uir_reg(UIR_REG_VIRT, v2); + in->saturate = 1; + in->src[0] = uir_swz(uir_reg(UIR_REG_VIRT, v0), UIR_SWZ(1, 2, 0, 3)); + in->src[1] = uir_neg(uir_swz(uir_reg(UIR_REG_VIRT, v1), UIR_SWZ_WWWW)); + in->src[1].absolute = 1; + in->nsrc = 2; + + in = uir_emit(s, UIR_DP4); + in->dst = uir_reg(UIR_REG_OUT, 0); + in->writemask = 0x1; + in->src[0] = uir_reg(UIR_REG_VIRT, v2); + in->src[1] = uir_imm4(0.1f, -0.0f, 1e-30f, 3.4e38f); + in->nsrc = 2; + in->comment = "project"; + + in = uir_emit(s, UIR_EMIT); + in = uir_emit(s, UIR_RET); + (void)in; + return s; +} + +static struct uir_shader *build_fragment(void) +{ + struct uir_shader *s = uir_shader_new(UIR_STAGE_FRAGMENT); + struct uir_binding b; + struct uir_insn *in; + uint32_t lbl; + + s->uses_kill = 1; + memset(&b, 0, sizeof b); + b.cls = UIR_REG_TEXCOORD; b.index = 0; b.slot = 0; b.components = 4; + b.name = "vTexCoord"; + uir_add_binding(s, &b); + b.cls = UIR_REG_SAMPLER; b.index = 0; b.slot = 0; b.components = 1; + b.name = "uTex"; + uir_add_binding(s, &b); + b.cls = UIR_REG_IMM; b.index = 0; b.slot = 0; b.components = 4; + b.name = NULL; + uir_add_binding(s, &b); + + s->nvirt = 4; + lbl = uir_alloc_label(s); + + in = uir_emit(s, UIR_TEX); + in->type = UIR_I8; + in->dst = uir_reg(UIR_REG_VIRT, 0); + in->src[0] = uir_reg(UIR_REG_TEXCOORD, 0); + in->src[1] = uir_reg(UIR_REG_SAMPLER, 0); + in->nsrc = 2; + + in = uir_emit(s, UIR_TEXPROJ); + in->dst = uir_reg(UIR_REG_VIRT, 1); + in->src[0] = uir_reg(UIR_REG_TEXCOORD, 0); + in->src[1] = uir_reg(UIR_REG_SAMPLER, 0); + in->nsrc = 2; + + in = uir_emit(s, UIR_TEXLOD); + in->dst = uir_reg(UIR_REG_VIRT, 2); + in->src[0] = uir_reg(UIR_REG_TEXCOORD, 0); + in->src[1] = uir_reg(UIR_REG_SAMPLER, 0); + in->src[2] = uir_imm1(0.5f); + in->nsrc = 3; + in->comment = "explicit lod"; + + in = uir_emit(s, UIR_BRC); + in->src[0] = uir_swz(uir_reg(UIR_REG_VIRT, 0), UIR_SWZ_WWWW); + in->nsrc = 1; + in->branch_target = lbl; + + in = uir_emit(s, UIR_KILL); + in->src[0] = uir_neg(uir_reg(UIR_REG_VIRT, 1)); + in->nsrc = 1; + + in = uir_emit(s, UIR_LABEL); + in->label_id = lbl; + in->comment = "skip the kill"; + + in = uir_emit(s, UIR_MUL); + in->type = UIR_I10; + in->saturate = 1; + in->dst = uir_reg(UIR_REG_VIRT, 3); + in->writemask = 0; + in->src[0] = uir_reg(UIR_REG_VIRT, 0); + in->src[1] = uir_reg(UIR_REG_VIRT, 2); + in->nsrc = 2; + + in = uir_emit(s, UIR_MOV); + in->dst = uir_reg(UIR_REG_OUT, 0); + in->src[0] = uir_reg(UIR_REG_VIRT, 3); + in->nsrc = 1; + + in = uir_emit(s, UIR_EMIT); + in = uir_emit(s, UIR_RET); + (void)in; + return s; +} + +static struct uir_shader *build_torture(void) +{ + static const float immv[8] = { + 0.1f, -0.0f, 1e-30f, 3.4e38f, -1.5f, 0.0f, 1e30f, -3.4e38f + }; + static const enum uir_regclass scls[5] = { + UIR_REG_VIRT, UIR_REG_IN, UIR_REG_UNIFORM, UIR_REG_TEXCOORD, + UIR_REG_IMM + }; + struct uir_shader *s = uir_shader_new(UIR_STAGE_FRAGMENT); + struct uir_binding b; + struct uir_insn *in; + int op, c, sw, na; + + s->uses_kill = 1; + s->nvirt = 32; + s->next_label = 7; + + memset(&b, 0, sizeof b); + for (c = 0; c <= UIR_REG_SAMPLER; c++) { + b.cls = (enum uir_regclass)c; + b.index = (uint32_t)c; + b.slot = (uint32_t)(c * 4 + 1); + b.components = (uint8_t)(c % 4 + 1); + b.name = (c & 1) ? NULL : "binding \\ \"name\""; + uir_add_binding(s, &b); + } + + for (op = 0; op < UIR_OP_COUNT; op++) { + unsigned j, n; + + in = uir_emit(s, (enum uir_op)op); + if (op == UIR_LABEL) { + in->label_id = 3; + in->comment = "the only label"; + continue; + } + in->type = (enum uir_type)(op % 5); + in->saturate = (uint8_t)(op % 3 == 0); + if (op_has_dst(in->op)) { + in->dst = uir_reg((op & 1) ? UIR_REG_VIRT : UIR_REG_OUT, + (uint32_t)(op % 8)); + in->dst.swizzle = (uint8_t)(op * 7); + in->writemask = (uint8_t)(op & 0xf); + } + n = op_nsrc[op]; + in->nsrc = n; + for (j = 0; j < n; j++) { + int k = (op + (int)j) % 5; + + if (op_is_tex(in->op) && j == 1) { + in->src[j] = uir_reg(UIR_REG_SAMPLER, j); + continue; + } + if (scls[k] == UIR_REG_IMM) { + int o = (op + (int)j) % 5; + + in->src[j] = uir_imm4(immv[o], immv[o + 1], + immv[o + 2], immv[o + 3]); + } else { + in->src[j] = uir_reg(scls[k], (uint32_t)(op + (int)j)); + } + in->src[j].swizzle = (uint8_t)(op * 11 + (int)j * 3); + in->src[j].negate = (uint8_t)((op + (int)j) & 1); + in->src[j].absolute = (uint8_t)(((op + (int)j) >> 1) & 1); + } + if (in->op == UIR_BR || in->op == UIR_BRC) + in->branch_target = 3; + if (op % 4 == 0) + in->comment = "op comment"; + } + + /* every swizzle against every negate/absolute combination */ + for (sw = 0; sw < 256; sw++) + for (na = 0; na < 4; na++) { + in = uir_emit(s, UIR_MOV); + in->dst = uir_reg(UIR_REG_VIRT, (uint32_t)(sw % 32)); + in->src[0] = uir_swz(uir_reg(UIR_REG_UNIFORM, (uint32_t)sw), + (uint8_t)sw); + in->src[0].negate = (uint8_t)(na & 1); + in->src[0].absolute = (uint8_t)((na >> 1) & 1); + in->nsrc = 1; + } + return s; +} + +static void check_validate(void) +{ + struct uir_shader *s; + char err[128]; + int bad = 0; + + s = build_vertex(); + if (uir_validate(s, err, sizeof err) != 0) { + printf(" vertex should validate: %s\n", err); + bad++; + } + uir_shader_free(s); + + s = build_fragment(); + if (uir_validate(s, err, sizeof err) != 0) { + printf(" fragment should validate: %s\n", err); + bad++; + } + uir_shader_free(s); + + s = uir_shader_new(UIR_STAGE_VERTEX); + s->nvirt = 2; + uir_emit(s, UIR_MOV)->nsrc = 0; + if (uir_validate(s, err, sizeof err) == 0) { + printf(" wrong source count not caught\n"); + bad++; + } + uir_shader_free(s); + + s = uir_shader_new(UIR_STAGE_VERTEX); + s->nvirt = 2; + { + struct uir_insn *in = uir_emit(s, UIR_MOV); + + in->dst = uir_reg(UIR_REG_VIRT, 0); + in->src[0] = uir_reg(UIR_REG_VIRT, 1); + in->nsrc = 1; + } + if (uir_validate(s, err, sizeof err) == 0) { + printf(" undefined read not caught\n"); + bad++; + } + uir_shader_free(s); + + s = uir_shader_new(UIR_STAGE_VERTEX); + uir_emit(s, UIR_BR)->branch_target = 9; + if (uir_validate(s, err, sizeof err) == 0) { + printf(" missing branch label not caught\n"); + bad++; + } + uir_shader_free(s); + + s = uir_shader_new(UIR_STAGE_VERTEX); + uir_emit(s, UIR_LABEL)->label_id = 1; + uir_emit(s, UIR_LABEL)->label_id = 1; + if (uir_validate(s, err, sizeof err) == 0) { + printf(" duplicate label not caught\n"); + bad++; + } + uir_shader_free(s); + + s = uir_shader_new(UIR_STAGE_FRAGMENT); + s->nvirt = 1; + { + struct uir_insn *in = uir_emit(s, UIR_ADD); + + in->dst = uir_reg(UIR_REG_VIRT, 0); + in->src[0] = uir_reg(UIR_REG_SAMPLER, 0); + in->src[1] = uir_reg(UIR_REG_IN, 0); + in->nsrc = 2; + } + if (uir_validate(s, err, sizeof err) == 0) { + printf(" misplaced sampler not caught\n"); + bad++; + } + uir_shader_free(s); + + if (bad) + nfail++; + printf("%s: validate\n", bad ? "FAIL" : "PASS"); +} + +static void check_parse_errors(void) +{ + static const char *const bad[] = { + "stage sideways\n", + " mov v0, v1\n", + " frobnicate.f32 v0, v1\n", + " mov.f32 v0 v1\n", + " mov.f32 v0, #(1,2,3)\n", + " mov.f32 v0, v1.q\n", + "bind bogus 0 slot 0 comp 4\n", + "bind in 0 slot 0 comp 4 name \"unterminated\n", + "label x:\n", + " mov.f32.xq v0, v1\n" + }; + size_t i; + int fails = 0; + + for (i = 0; i < sizeof bad / sizeof bad[0]; i++) { + char err[128] = ""; + struct uir_shader *s = uir_parse(bad[i], err, sizeof err); + + if (s) { + printf(" input %lu should not parse\n", (unsigned long)i); + uir_shader_free(s); + fails++; + } else if (strncmp(err, "line ", 5)) { + printf(" input %lu error lacks line number: %s\n", + (unsigned long)i, err); + fails++; + } + } + if (fails) + nfail++; + printf("%s: parse errors\n", fails ? "FAIL" : "PASS"); +} + +int main(void) +{ + struct uir_shader *s; + + s = build_vertex(); + roundtrip(s, "vertex"); + uir_shader_free(s); + + s = build_fragment(); + roundtrip(s, "fragment"); + uir_shader_free(s); + + s = build_torture(); + roundtrip(s, "torture"); + uir_shader_free(s); + + check_validate(); + check_parse_errors(); + + printf("%s: %d failing check%s\n", nfail ? "FAILED" : "OK", nfail, + nfail == 1 ? "" : "s"); + return nfail != 0; +} + +#endif /* UIR_IR_SELFTEST */ diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/usse_ir.h mesa-26.2.2/src/gallium/drivers/sgx/usse_ir.h --- mesa-26.2.2.orig/src/gallium/drivers/sgx/usse_ir.h 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/usse_ir.h 2026-09-08 10:57:36.682321603 +0200 @@ -0,0 +1,439 @@ +/* + * usse_ir.h - shared IR contract for the SGX535 USSE compiler + * + * This header is the interface between the three halves of the compiler: + * the GLSL ES front end and the fixed-function generator both PRODUCE it, + * the backend CONSUMES it. It is deliberately fixed: change it only by + * agreement, because three independent implementations depend on it. + * + * Design notes: + * - Vector-first. SGX535 is a 4-wide float unit with per-channel write + * masks and source swizzles, so the IR keeps vectors intact rather than + * scalarising and asking the backend to re-vectorise. + * - Virtual registers, unbounded. The backend allocates to the real banks + * (temporary r, primary attribute pa, secondary attribute sa, output o). + * - Not SSA. Straight-line GL shaders gain little from it and the backend + * needs live ranges, not phi nodes. Blocks exist for control flow only. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#ifndef USSE_IR_H +#define USSE_IR_H + +#include +#include + +#define UIR_MAX_SRC 4 +/* Uniform vec4s a program may name. */ +#define UIR_MAX_UNIFORM 64 + +/* Register classes. The backend maps UIR_REG_VIRT onto real banks; the other + * classes are fixed by the hardware ABI and must not be reallocated. */ +enum uir_regclass { + UIR_REG_VIRT = 0, /* virtual, backend-allocated */ + UIR_REG_IN, /* shader input: pa bank (iterated/attribute)*/ + UIR_REG_UNIFORM, /* constant: sa bank, PDS-loaded */ + UIR_REG_OUT, /* shader output: position / colour */ + UIR_REG_IMM, /* literal, see uir_ref.imm */ + UIR_REG_TEXCOORD, /* iterated texture coordinate set */ + UIR_REG_SAMPLER /* texture unit index, SMP operand only */ +}; + +/* Scalar type of a value. SGX535 has native f32 plus packed integer formats; + * GLSL ES 1.00 needs only float/int/bool, but the packed types matter for + * texture results and for the blend/combine chain, which runs in i8. */ +enum uir_type { + UIR_F32 = 0, + UIR_I32, + UIR_I16, + UIR_I8, /* the fixed-function combine chain works here */ + UIR_I10 +}; + +/* One operand. A swizzle is 8 bits, 2 per channel, selecting src channel + * 0..3 for dst channel x,y,z,w -- identity is 0xe4 (w=3,z=2,y=1,x=0). */ +struct uir_ref { + enum uir_regclass cls; + uint32_t index; /* register number within the class */ + uint8_t swizzle; /* 0xe4 = .xyzw identity */ + uint8_t negate; /* apply unary minus after swizzle */ + uint8_t absolute; /* apply abs() before negate */ + float imm[4]; /* valid only when cls == UIR_REG_IMM */ +}; + +enum uir_op { + UIR_NOP = 0, + UIR_MOV, + UIR_ADD, UIR_SUB, UIR_MUL, UIR_MAD, /* MAD = a*b + c */ + UIR_MIN, UIR_MAX, + UIR_FRC, UIR_FLOOR, + UIR_RCP, UIR_RSQ, UIR_LOG2, UIR_EXP2, UIR_SQRT, + UIR_DP3, UIR_DP4, /* replicate to dst mask */ + UIR_CROSS, UIR_NRM, + UIR_SLT, UIR_SGE, UIR_SEQ, UIR_SNE, /* 1.0 / 0.0 result */ + UIR_CMP, /* src0 < 0 ? src1 : src2 */ + UIR_TEX, /* sample: src0 = coord, src1 = UIR_REG_SAMPLER */ + UIR_TEXPROJ, /* as TEX, coord divided by .w */ + UIR_TEXLOD, /* as TEX, src2.x = explicit LOD */ + UIR_KILL, /* discard if src0 < 0 (alpha test, discard) */ + UIR_EMIT, /* emit the shader output; backend places it */ + UIR_BR, /* unconditional, target = branch_target */ + UIR_BRC, /* branch if src0 != 0 */ + UIR_LABEL, + UIR_RET, + /* Screen-space derivatives. The part has them - DSX and DSY, group + * 0x04 - and Imagination's own compiler emits "fdsx r0, r0, r0" for + * dFdx, both sources the same register, which is the form below. */ + UIR_DDX, UIR_DDY, + /* The face the fragment belongs to, as a float: +1.0 where bit 0 of + * the USE's BFCONTROL register is clear, -1.0 where it is set. Read + * the way the vendor's compiler reads it (usc2/icvt_f32.c + * UF_MISC_FACETYPE), so it needs no varying; which winding sets the + * bit is the driver's to say (sgx_face_swap_of(), measured). */ + UIR_FACE, + /* As TEX, src2.x = a LOD bias the unit adds to the level it computed + * (the SMP's LODM_BIAS mode; TEXLOD is LODM_REPLACE). */ + UIR_TEXBIAS, + UIR_OP_COUNT +}; + +struct uir_insn { + enum uir_op op; + enum uir_type type; /* precision the op runs at */ + struct uir_ref dst; + uint8_t writemask; /* bit0=x .. bit3=w; 0xf = all */ + uint8_t saturate; /* clamp result to [0,1] */ + struct uir_ref src[UIR_MAX_SRC]; + unsigned nsrc; + uint32_t branch_target; /* label id for UIR_BR / UIR_BRC */ + uint32_t label_id; /* for UIR_LABEL */ + /* Coordinate components a UIR_TEX* reads: 3 for a volume or a + * cube, else two (zero is two, so an emitted sample need not say). */ + uint8_t tex_dim; + const char *comment; /* optional, carried into asm output */ +}; + +enum uir_stage { + UIR_STAGE_VERTEX = 0, + UIR_STAGE_FRAGMENT +}; + +/* Declares what the shader expects the driver to have set up: which pa + * registers hold which iterated value, which sa registers hold which + * uniform. The driver needs this to build the PDS program, so it is part + * of the compiler's output, not a side note. */ +struct uir_binding { + enum uir_regclass cls; + uint32_t index; /* register index within the class */ + uint32_t slot; /* attribute slot / uniform offset */ + uint8_t components; /* 1..4 */ + const char *name; /* may be NULL for fixed-function */ +}; + +struct uir_shader { + enum uir_stage stage; + struct uir_insn *insns; + size_t ninsns; + size_t insns_cap; + + struct uir_binding *bindings; + size_t nbindings; + size_t bindings_cap; + + uint32_t nvirt; /* number of virtual registers used */ + uint32_t next_label; + + /* Fragment only: set if the shader can discard, which forces the + * backend to keep the depth write until after the kill. */ + uint8_t uses_kill; +}; + +/* --- construction helpers; implemented in usse_ir.c ------------------- */ + +struct uir_shader *uir_shader_new(enum uir_stage stage); +void uir_shader_free(struct uir_shader *s); +struct uir_insn *uir_emit(struct uir_shader *s, enum uir_op op); +uint32_t uir_alloc_virt(struct uir_shader *s); +uint32_t uir_alloc_label(struct uir_shader *s); +int uir_add_binding(struct uir_shader *s, + const struct uir_binding *b); + +/* Reference builders. */ +struct uir_ref uir_reg(enum uir_regclass cls, uint32_t index); +struct uir_ref uir_imm4(float x, float y, float z, float w); +struct uir_ref uir_imm1(float v); +struct uir_ref uir_swz(struct uir_ref r, uint8_t swizzle); +struct uir_ref uir_neg(struct uir_ref r); + +#define UIR_SWZ(x, y, z, w) ((uint8_t)((x) | ((y) << 2) | ((z) << 4) | ((w) << 6))) +#define UIR_SWZ_XYZW UIR_SWZ(0, 1, 2, 3) /* 0xe4 */ +#define UIR_SWZ_XXXX UIR_SWZ(0, 0, 0, 0) +#define UIR_SWZ_WWWW UIR_SWZ(3, 3, 3, 3) + +/* Text form, for tests and for eyeballing front-end output. Round-trips: + * uir_print then uir_parse must produce an identical shader. */ +int uir_print(const struct uir_shader *s, char *buf, size_t n); +struct uir_shader *uir_parse(const char *text, char *err, size_t errn); + +/* Structural validation: operand counts, mask/swizzle sanity, undefined + * register reads, branches to missing labels. Returns 0 on success. */ +int uir_validate(const struct uir_shader *s, + char *err, size_t errn); + +/* --- backend entry point ---------------------------------------------- */ + +struct uir_codegen_opts { + int optimise; /* 0 = none, 1 = peephole, 2 = full */ + int max_temps; /* hardware temp budget; 0 = default */ + int allow_dual_issue; + /* The fragment output is already in the render target's format, so + * ending the program is a move rather than a float-to-unorm8 pack. + * Which applies depends on how the attributes are delivered: the + * captured frame hands the fragment program a packed 8888 dword in a + * primary attribute, and packing that again produces nonsense. */ + int frag_out_packed; + /* The value came out of the record's packed colour, which the MTE + * packs in the opposite byte order to the one the pack path writes + * for the target - so this program's pack runs reversed. Only a + * pass-through of that colour sets it; everything else, including + * every program that computes, keeps the ordinary order. */ + int frag_out_reverse; + /* The texel is not sampled by the program. On the frame this driver + * builds, the texture unit samples per fragment and the result is + * delivered as a primary attribute - pa0 is the iterated colour and + * pa[1 + u] is unit u's texel, measured on a two-unit frame. A TEX is + * then a read of that register rather than an SMP, and the coordinate + * is the iterated one, so a computed coordinate has no path and is + * refused rather than sampled at the wrong place. + * + * Zero keeps the general form: the program issues SMP itself. */ + /* Pack the pixel red first. The record a hardware vertex program + * writes carries the colour in the order the shader computed it, + * where the draw module's record carries it blue first. */ + int frag_out_rgba; + int frag_tex_preiterated; + /* Which primary attribute unit 0's texel arrives in. One, measured on + * the captured frame; the field exists so the driver can sweep it on + * hardware rather than the number being written in twice. */ + int frag_tex_pa_base; + /* Inputs delivered as one packed 8888 dword rather than four floats, + * by index. The frame hands a varying over that way for a good deal + * less than it costs to iterate it as a coordinate set - measured, the + * same scene renders in 5 ms against 106 - but a program that computes + * with the value needs floats, so each one is unpacked into a quad of + * temporaries at entry and read from there. */ + unsigned frag_in_packed; + /* Inputs the iterator delivers as half floats rather than as one + * register a component, by index. The DOUTI's USEFORMAT field says + * F16, two components share a register, and the value keeps a float's + * range - so unlike a packed 8888 varying this one can carry a + * computed quantity. The unpack is the same shape: a quad of + * temporaries at entry, read from there. An index may not be in both + * this and frag_in_packed. */ + unsigned frag_in_f16; + /* The record's colour slot is packed low byte first for a varying the + * draw module reorders into it, and in the shader's own order for one + * the frame iterates onto its own DOUTI. The unpack has to follow. */ + int frag_packed_rgba; + /* Write the colour output in the order the record's colour slot is + * packed from - blue first, the draw module's order. The slot stays + * four floats: the MTE reads the base colour as F32 x4 and the + * iterator packs it into the one dword a packed-colour issue + * delivers. Set only for a transform paired with a fragment program + * that takes a packed colour. */ + int vtx_pack_colour; + /* Where each fragment input sits in the primary attribute bank. The + * default is four registers per input, which is right when every + * input is an iterated vector; it is wrong the moment the frame + * delivers a packed colour (one register) or a sampled texel (one) + * alongside them, so the driver hands the real bases in. Zero entries + * with frag_in_bases clear keep the default. */ + unsigned char frag_in_base[16]; + int frag_in_bases; + /* Where a vertex program's uniforms live. The fragment stage reads + * them out of the secondary attribute bank, which a secondary PDS + * program fills; the vertex stage has no such program in the captured + * frame, and the stock driver instead has the vertex PDS DMA them + * into the primary attributes just past the vertex record. Non-zero + * puts uniform i at pa[base + 4*i]; zero keeps the sa form. */ + int vtx_uniform_pa_base; + /* Where each vertex output goes. The record the fragment stage + * iterates has fixed slots - position, colour, coordinate - and a + * shader's outputs are numbered in declaration order, which is not + * the same thing. Output i is emitted to o[4 * vtx_out_slot[i]] when + * vtx_out_slots is non-zero; otherwise output i goes to o[4 * i]. */ + /* A vertex program has no literal pool it can reach: the pool lives in + * the secondary attribute bank and this stage has nothing that fills + * it. Set for any vertex program, whatever bank its uniforms are in, + * so that an immediate is resolved to the hardware's own 0.0 and 1.0 + * or refused - never placed in a pool the driver would then have to + * upload somewhere. */ + int vtx_no_literal_pool; + /* Materialise a vertex program's uniforms as LIMM immediates in the + * instruction stream rather than reading them from an attribute bank. + * The primary bank is 32 registers deep and the vertex record is in + * front of them, which caps a program at eighteen uniform dwords; the + * secondary bank is not reachable on this stage. An immediate needs no + * register at all, and the driver rewrites the program every flush + * anyway, so it can patch the values in without recompiling - the + * result reports where they sit. */ + int vtx_uniform_limm; + /* How many output slots the MTE emits per vertex, which is the + * record's width in quads. The backend fills the ones the program + * does not write: the part emits them whatever they hold. Zero means + * the caller does not know and only the gaps below the highest + * written slot are filled. */ + unsigned vtx_out_slots_emitted; + unsigned char vtx_out_slot[16]; + int vtx_out_slots; + /* Where each vertex output starts, in emitted dwords. The slot map + * above puts output i at 4 * slot, which is a quad per slot; the + * iterator instead packs the coordinate sets by their own widths + * behind the eight fixed dwords, so the two only agree while every + * set is four wide. This carries the packed offset so they agree for + * any width. Used when vtx_out_dwords is non-zero. */ + unsigned char vtx_out_dw[16]; + int vtx_out_dwords; + /* What each sampler unit's fetch delivers, so the backend can turn it + * into four floats. The unit returns one register per plane in the + * surface's own form: four bytes for the 8-bit formats (and 4444, + * 1555, 565, ETC1, which the TAG expands), one float for F32, one or + * two halves for F16 and F1616, and likewise the 16-bit normalised + * forms. A format wider than a register is stored as several planes + * ("chunks", the DDK's word) each described by a state block of its + * own, sampled separately - sgx535pixfmts.h has RGBA32F as four F32 + * chunks and RGBA16F as two F1616. Both arrays hold ntex_units + * entries and are the caller's; unit u past the end, or a NULL + * array, is U8888 in one chunk. */ +#define UIR_TEXCLASS_U8888 0 +#define UIR_TEXCLASS_F32 1 +#define UIR_TEXCLASS_F16 2 +#define UIR_TEXCLASS_F1616 3 +#define UIR_TEXCLASS_U16 4 +#define UIR_TEXCLASS_S16 5 +#define UIR_TEXCLASS_U1616 6 +#define UIR_TEXCLASS_S1616 7 + const unsigned char *tex_class; + const unsigned char *tex_chunks; + unsigned ntex_units; + /* Fuse pairs of single-iteration multiply-adds into EFO instructions. + * Zero follows SGX_EFO in the environment, which is how the driver + * reaches it and which is off unless it is set; 1 turns the pass on + * and -1 off whatever the environment says, so a test can compile the + * same shader both ways in one process. */ + int efo; +}; + +#define UIR_MAX_LIMM 128u + +/* Where a uniform's four LIMM instructions sit, when they were materialised + * into the instruction stream: instruction `insn` carries uniform dword + * `dword`. The driver patches each one's immediate as the values change. */ +struct uir_limm_ref { + uint32_t insn, dword; +}; + +/* Literal pool. A compiled shader does not carry its constants inline: it + * reads them out of the secondary attribute bank, and the driver has to have + * put them there before the program runs (a secondary PDS program does it, + * see work/uniform-abi/). One entry is one vec4 occupying sa_count + * consecutive sa registers starting at sa_base; bits[] is the value in the + * exact dword order the driver uploads. */ +#define UIR_MAX_POOL 64 + +/* The secondary attribute bank, in 32-bit registers: uniforms, the per-sampler + * state blocks and the literal pool all come out of it + * (EURASIA_USE_SECATTR_BANK_SIZE). */ +#define UIR_SA_BANK 128 + +struct uir_literal { + float v[4]; /* the constant, as the compiler saw it */ + uint32_t bits[4]; /* v[] as raw IEEE-754 dwords, low first */ + uint32_t sa_base; /* first secondary attribute register */ + uint32_t sa_count; /* registers occupied; 4 in this backend */ +}; + +/* Emits 64-bit USSE instructions into out[], returns the count written, + * or negative on failure. The backend also fills the PDS-relevant register + * counts the driver needs to build its data segment, and the literal pool + * the driver must upload into the secondary attributes. */ +struct uir_codegen_result { + uint32_t ntemps; /* r registers used */ + /* Where a fragment program's second colour ended up, for a dual + * source blend to name. A fragment output is a temporary here - the + * pixel is written by the epilogue - so the second one is simply + * another temporary, and it never reaches the pixel back end at all: + * the blend reads it as SOP3's src0. has_src1 is zero for the usual + * program, which declares one output. */ + uint32_t src1_reg; + uint32_t has_src1; + uint32_t nprimattr; /* pa registers read */ + uint32_t nsecattr; /* sa registers the driver must provide */ + uint32_t ninsns; + + /* The constant pool. Deduplicated, allocated above the uniforms and + * the per-sampler state blocks, so pool_base is where the driver's + * upload starts and pool_dwords is how much of it there is. */ + struct uir_literal pool[UIR_MAX_POOL]; + uint32_t npool; /* entries in pool[] */ + uint32_t pool_base; /* sa register of pool[0] */ + /* Where the per-sampler state blocks start, which is where the SMP + * instructions this codegen emitted read them from. The driver cannot + * infer it: nuniform is the count the program declares and it + * under-reports, so a descriptor placed from it landed on a uniform + * the shader samples with and the sample read zeros from the block + * the code actually names. */ + uint32_t smp_base; /* sa register of sampler 0's block */ + /* One state block per chunk, unit-major: sampler u's chunk k reads + * block smp_slot[u] + k, at smp_base + 3 * that. nsamp entries, + * released by uir_codegen_result_fini(); nsmp_slots is the total. */ + unsigned char *smp_slot; + uint32_t nsamp, nsmp_slots; + uint32_t pool_dwords; /* dwords to upload, 4 * npool */ + /* Uniforms the program reads, in quads. A fragment program's uniform + * i is sa[4 * i], below the sampler blocks and the pool, so the driver + * has to upload this much before them. */ + uint32_t nuniform; + /* Where each uniform sits in the secondary attribute bank and how many + * registers it takes. Uniforms are packed to the components they are + * read through rather than a quad apiece, so the driver has to gather + * them into this layout instead of copying the block. Both hold + * nuniform entries and are released by uir_codegen_result_fini(). */ + uint8_t *uni_base; + uint8_t *uni_width; + + /* Uniforms materialised as immediates, when vtx_uniform_limm was on. + * They are emitted first and in order, so instruction 4 * i + c + * carries uniform dword 4 * i + c, but the map is reported rather + * than assumed. */ + struct uir_limm_ref limm[UIR_MAX_LIMM]; + uint32_t nlimm; + + /* The program discards, and ends by selecting between the pixel it + * computed and o0's incoming value. The driver has to draw it with the + * read-modify-write object type so that o0 arrives holding the + * destination. */ + uint32_t uses_kill; + + /* The program samples with a coordinate it packed to f16 in a + * temporary. The unit ignores that operand and stays at texel + * (0,0), so every pixel of the draw comes back one colour + * (work/smp-reverse README section 4). A computed coordinate goes + * as f32 registers now, so only SGX_SMP_F16_COORD asks for this; + * the flag stays so the driver can say what the escape costs. */ + uint32_t coord_f16; +}; + +/* Releases what a result carries. Safe on a zeroed result and on one from a + * failed codegen, and idempotent. */ +void uir_codegen_result_fini(struct uir_codegen_result *res); + +int uir_codegen(const struct uir_shader *s, + const struct uir_codegen_opts *opts, + uint64_t *out, size_t out_cap, + struct uir_codegen_result *res, + char *err, size_t errn); + +#endif /* USSE_IR_H */ diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/usse_isa.c mesa-26.2.2/src/gallium/drivers/sgx/usse_isa.c --- mesa-26.2.2.orig/src/gallium/drivers/sgx/usse_isa.c 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/usse_isa.c 2026-09-08 10:57:36.682794261 +0200 @@ -0,0 +1,1302 @@ +/* + * usse_isa.c - SGX535 USSE instruction encoder, decoder and printer + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#include +#include +#include + +#include "usse_isa.h" + +#define GRP(w1) (((w1) >> 27) & 0x1F) + +/* The destination and source bank numberings agree except that the vendor + * decoder swaps slots 3 and 4 between them. */ +static const uint8_t src_bank_code[USSE_BANK_COUNT] = { 0, 1, 2, 3, 5, 6, 7, 4 }; +static const uint8_t dst_bank_code[USSE_BANK_COUNT] = { 0, 1, 2, 4, 5, 0xFF, 7, 3 }; +static const uint8_t src_bank_from[8] = { USSE_TEMP, USSE_OUT, USSE_PA, USSE_SA, + USSE_INDEX, USSE_CONST, USSE_IMMB, + USSE_INTERNAL }; +static const uint8_t dst_bank_from[8] = { USSE_TEMP, USSE_OUT, USSE_PA, USSE_INDEX, + USSE_SA, USSE_CONST, USSE_INTERNAL, + USSE_INTERNAL }; +/* SMP takes its coordinate from an extended src0 bank with its own order. */ +static const uint8_t smp_s0_code[USSE_BANK_COUNT] = { 0, 2, 1, 3, 0xFF, 0xFF, 0xFF, 0xFF }; + +static const char *const bank_pfx[USSE_BANK_COUNT] = { + "r", "o", "pa", "sa", "c", "#", "i", "r[il + #" +}; + +static const uint8_t smp_s0_from[4] = { USSE_TEMP, USSE_PA, USSE_OUT, USSE_SA }; + +/* Which source slots each modelled opcode reads. Unsized on purpose: with an + * explicit [USSE_OP_COUNT] the assert below is a tautology, and that is how + * both tables came to be two entries short of the enum - DSX and DSY were + * added to the opcode list and not here - so every opcode from DSX on read + * its neighbour's row. A DSX encoded without its second source, a sample + * whose coordinate and state registers were never counted, and a listing + * that rendered a branch as "wdf" were all that one gap. */ +static const uint8_t op_srcs[] = { + 0x7, 0x7, 0x7, 0x6, /* MAD ADM MSA SUBFLR */ + 0x2, 0x2, 0x2, 0x2, /* RCP RSQ LOG EXP */ + 0x6, /* DP */ + 0x6, 0x6, /* MIN MAX */ + 0x6, 0x6, /* DSX DSY */ + 0x2, 0x7, /* MOV MOVC */ + 0x6, 0x6, /* PCKU8F32 UNPCKF32U8 */ + 0x3, /* SMP: src0 + src1 */ + 0x0, 0x0, 0x0, 0x0, 0x0, 0x0, /* LIMM SMLSI EMIT NOP BA BR */ + 0x6, 0x0, /* TEST WDF */ + 0x6, /* PCKF16F32 */ + 0x6, 0x6, 0x6, /* UNPCKF32F16 U16 S16 */ + 0x7 /* EFO */ +}; + +_Static_assert(sizeof op_srcs / sizeof op_srcs[0] == USSE_OP_COUNT, + "op_srcs[] must have one entry per opcode"); + +static const char *const op_names[] = { + "fmad", "fadm", "fmsa", "fsubflr", + "frcp", "frsq", "flog", "fexp", + "fdp", "fmin", "fmax", + "fdsx", "fdsy", + "mov", "movc", "pcku8f32", "unpckf32u8", "smp", "mov", "smlsi", "emit", "nop", + "ba", "br", "fadd", "wdf", "pckf16f32", + "unpckf32f16", "unpckf32u16", "unpckf32s16", + "efo" +}; +_Static_assert(sizeof op_names / sizeof op_names[0] == USSE_OP_COUNT, + "op_names[] must have one entry per opcode"); + +static int is_pck(unsigned op) +{ + return op == USSE_PCKU8F32 || op == USSE_UNPCKF32U8 || + op == USSE_PCKF16F32 || op == USSE_UNPCKF32F16 || + op == USSE_UNPCKF32U16 || op == USSE_UNPCKF32S16; +} + +/* The PCK source-format field for the unpacks whose destination is f32. */ +static unsigned unpck_src_fmt(unsigned op) +{ + switch (op) { + case USSE_UNPCKF32F16: return 5; + case USSE_UNPCKF32U16: return 3; + case USSE_UNPCKF32S16: return 4; + default: return 0; /* u8 */ + } +} + +const char *usse_op_name(unsigned op) +{ + return op < USSE_OP_COUNT ? op_names[op] : "?"; +} + +unsigned usse_op_srcs(unsigned op) +{ + return op < USSE_OP_COUNT ? op_srcs[op] : 0; +} + +int usse_op_has_skipinv(unsigned op) +{ + switch (op) { + case USSE_SMLSI: case USSE_EMIT: case USSE_NOP: + case USSE_BA: case USSE_BR: case USSE_WDF: + return 0; + default: + return op < USSE_OP_COUNT; + } +} + +/* Only the arithmetic forms carry per-source negate and absolute bits; MOV, + * MOVC, the pack and the sample forms reuse those bits for other fields. */ +int usse_op_has_srcmod(unsigned op) +{ + switch (op) { + case USSE_MAD: case USSE_ADM: case USSE_MSA: case USSE_SUBFLR: + case USSE_RCP: case USSE_RSQ: case USSE_LOG: case USSE_EXP: + case USSE_DSX: case USSE_DSY: case USSE_DP: case USSE_MIN: case USSE_MAX: + case USSE_EFO: + return 1; + default: + return 0; + } +} + +/* Which optional fields each form actually has a home for. The encoder and + * the printer both consult this, so they cannot disagree. */ +#define F_MASK 0x01 +#define F_SYNCS 0x02 +#define F_NOSCH 0x04 +#define F_END 0x08 +#define F_PRED 0x10 + +/* Whether the opcode names a destination register. Not the same question as + * F_MASK: LIMM and TEST carry no write mask and still write one. */ +int usse_op_has_dst(unsigned op) +{ + switch (op) { + case USSE_MOV: case USSE_MOVC: + case USSE_MAD: case USSE_ADM: case USSE_MSA: case USSE_SUBFLR: + case USSE_RCP: case USSE_RSQ: case USSE_LOG: case USSE_EXP: + case USSE_DSX: case USSE_DSY: case USSE_DP: case USSE_MIN: case USSE_MAX: + case USSE_PCKU8F32: case USSE_UNPCKF32U8: case USSE_PCKF16F32: + case USSE_UNPCKF32F16: case USSE_UNPCKF32U16: case USSE_UNPCKF32S16: + case USSE_SMP: case USSE_LIMM: case USSE_TEST: case USSE_EFO: + return 1; + default: + return 0; + } +} + +static unsigned op_feat(unsigned op) +{ + switch (op) { + case USSE_MOV: case USSE_MOVC: + case USSE_MAD: case USSE_ADM: case USSE_MSA: case USSE_SUBFLR: + case USSE_RCP: case USSE_RSQ: case USSE_LOG: case USSE_EXP: + case USSE_DSX: case USSE_DSY: case USSE_DP: case USSE_MIN: case USSE_MAX: case USSE_PCKU8F32: + case USSE_UNPCKF32U8: case USSE_PCKF16F32: + case USSE_UNPCKF32F16: case USSE_UNPCKF32U16: case USSE_UNPCKF32S16: + return F_MASK | F_SYNCS | F_NOSCH | F_END | F_PRED; + case USSE_SMP: + return F_MASK | F_SYNCS | F_PRED; + case USSE_LIMM: + return F_NOSCH | F_END | F_PRED; + case USSE_EMIT: + return F_END; + case USSE_NOP: + return F_SYNCS | F_END; + case USSE_TEST: + return F_PRED; + /* useasm.c:7903 CheckFlags for an EFO admits only .skipinv, .nosched, + * a predicate and a repeat: w1[18] and w1[20] carry ISRC and DSRC, so + * the form has no .end and no .syncs, and w1[15:12] is the repeat + * count alone, so it has no write mask either. */ + case USSE_EFO: + return F_NOSCH | F_PRED; + case USSE_WDF: + return 0; + case USSE_BA: case USSE_BR: + return F_PRED; + default: + return 0; + } +} + +static int op_has_repeat(unsigned op) +{ + return op == USSE_EFO || (op_feat(op) & F_MASK) != 0; +} + +static void putf(uint32_t *w, unsigned hi, unsigned lo, uint32_t v) +{ + uint32_t m = (hi - lo == 31) ? 0xFFFFFFFFu : (((1u << (hi - lo + 1)) - 1) << lo); + *w = (*w & ~m) | ((v << lo) & m); +} + +static uint32_t getf(uint32_t w, unsigned hi, unsigned lo) +{ + return (w >> lo) & ((hi - lo == 31) ? 0xFFFFFFFFu : ((1u << (hi - lo + 1)) - 1)); +} + +/* Group and sub-selector for the plain ALU forms. */ +static int alu_group(unsigned op, unsigned *grp, unsigned *sub) +{ + switch (op) { + case USSE_MAD: *grp = 0x00; *sub = 0; return 0; + case USSE_ADM: *grp = 0x00; *sub = 1; return 0; + case USSE_MSA: *grp = 0x00; *sub = 2; return 0; + case USSE_SUBFLR: *grp = 0x00; *sub = 3; return 0; + case USSE_RCP: *grp = 0x01; *sub = 0; return 0; + case USSE_RSQ: *grp = 0x01; *sub = 1; return 0; + case USSE_LOG: *grp = 0x01; *sub = 2; return 0; + case USSE_EXP: *grp = 0x01; *sub = 3; return 0; + case USSE_DP: *grp = 0x02; *sub = 0; return 0; + case USSE_MIN: *grp = 0x03; *sub = 0; return 0; + /* Group 0x04, sub 0 and 1. Corroborated by Imagination's compiler + * emitting fdsx at exactly this encoding for dFdx. */ + case USSE_DSX: *grp = 0x04; *sub = 0; return 0; + case USSE_DSY: *grp = 0x04; *sub = 1; return 0; + case USSE_MAX: *grp = 0x03; *sub = 1; return 0; + default: return -1; + } +} + +/* Register numbers per bank. The field carries 0..127, but the top four + * temporaries are the internal registers i0..i3 under another name (useasm.c + * CheckAndEncodeRegisterNumber, EURASIA_USE_NUM_TEMPS_MAPPED_TO_FPI) and the + * constant bank has 64 entries (EURASIA_USE_FPCONSTANT_BANK_SIZE); a number + * past either selects something else without a word said. */ +static int chk_num(unsigned bank, unsigned num, const char *what, + const char **err) +{ + static char msg[96]; + unsigned max = 127; + + if (bank >= USSE_BANK_COUNT) { *err = "bad register bank"; return -1; } + if (bank == USSE_TEMP) + max = 123; + else if (bank == USSE_CONST) + max = 127; /* c0..c63, then the global bank as g0..g63 */ + if (num <= max) + return 0; + snprintf(msg, sizeof msg, "%s register %s%u is past the %u the bank " + "holds", what, bank_pfx[bank], num, max); + *err = msg; + return -1; +} + +/* Source 0 has no extended bank set, so an internal register there is one + * of the top four temporaries (useasm.c GPIRegistersUseTempBank: the + * dedicated bank exists only where the extended set does). */ +static int s0_num(const struct usse_operand *s, unsigned *num, const char **err) +{ + if (s->bank == USSE_INTERNAL) { + if (s->num > 3) { + *err = "source 0 reaches internal registers i0..i3 only"; + return -1; + } + *num = s->num + 124; + return 0; + } + if (s->bank != USSE_TEMP && s->bank != USSE_PA) { + *err = "src0 can only address the temporary, primary attribute or internal bank"; + return -1; + } + *num = s->num; + return 0; +} + +/* EFO reaches the internal registers through the top four temporaries in + * every operand slot, not through the extended bank: w1[19] carries ISRC + * where a destination would hold DBEXT, and useasm.c:7994-7997 encodes all + * three sources and the destination with the extension disabled. `which` is + * 0 for the destination and 1..3 for source 0..2. */ +static int efo_num(const struct usse_operand *o, unsigned which, unsigned *num, + unsigned *code, const char **err) +{ + static const uint8_t dst_ok[USSE_BANK_COUNT] = { 1, 1, 1, 0, 0, 0, 1, 1 }; + static const uint8_t s0_ok[USSE_BANK_COUNT] = { 1, 0, 1, 0, 0, 0, 1, 0 }; + static const uint8_t s12_ok[USSE_BANK_COUNT] = { 1, 1, 1, 1, 0, 0, 1, 0 }; + static const uint8_t dst_code[USSE_BANK_COUNT] = { 0, 1, 2, 0, 0, 0, 0, 3 }; + static const uint8_t s12_code[USSE_BANK_COUNT] = { 0, 1, 2, 3, 0, 0, 0, 0 }; + const uint8_t *ok = which == 0 ? dst_ok : (which == 1 ? s0_ok : s12_ok); + + if (o->bank >= USSE_BANK_COUNT || !ok[o->bank]) { + *err = which == 0 ? "efo destination bank" : + which == 1 ? "efo src0 takes the temporary, primary " + "attribute or internal bank only" + : "efo src1/src2 take the temporary, output, " + "primary attribute, secondary attribute or " + "internal bank only"; + return -1; + } + if (o->bank == USSE_INTERNAL) { + if (o->num > 3) { + *err = "internal register past i3"; + return -1; + } + *num = o->num + 124u; + *code = 0; + return 0; + } + if (chk_num(o->bank, o->num, which ? "efo source" : "efo destination", err)) + return -1; + *num = o->num; + *code = which == 0 ? dst_code[o->bank] : + which == 1 ? (o->bank == USSE_PA) : s12_code[o->bank]; + return 0; +} + +static int enc_efo(uint32_t *w0, uint32_t *w1, const struct usse_insn *in, + const char **err) +{ + unsigned num, code, i; + + if (in->end || in->syncstart) { + *err = "an efo carries no .end and no .syncs"; + return -1; + } + if (in->mask > 1) { + *err = "an efo carries no write mask"; + return -1; + } + if (in->repeat > USSE_EFO_MAX_REPEAT) { + *err = "an efo repeats at most four times"; + return -1; + } + if (in->efo_dsrc > 3 || in->efo_isrc > 3 || in->efo_asrc > 3 || + in->efo_msrc > 3) { + *err = "efo source select out of range"; + return -1; + } + putf(w1, 31, 27, 0x07); + if (efo_num(&in->dst, 0, &num, &code, err)) + return -1; + putf(w0, 27, 21, num); + putf(w1, 1, 0, code); + if (efo_num(&in->src[0], 1, &num, &code, err)) + return -1; + putf(w0, 20, 14, num); + putf(w1, 2, 2, code); + if (efo_num(&in->src[1], 2, &num, &code, err)) + return -1; + putf(w0, 13, 7, num); + putf(w0, 31, 30, code); + if (efo_num(&in->src[2], 3, &num, &code, err)) + return -1; + putf(w0, 6, 0, num); + putf(w0, 29, 28, code); + for (i = 0; i < 3; i++) { + unsigned mod = (unsigned)in->src[i].neg | (in->src[i].abs << 1); + + putf(w1, 8 - 2 * i, 7 - 2 * i, mod); + } + putf(w1, 21, 20, in->efo_dsrc); + putf(w1, 19, 18, in->efo_isrc); + putf(w1, 17, 16, in->efo_asrc); + putf(w1, 15, 14, in->efo_msrc); + if (in->repeat) + putf(w1, 13, 12, in->repeat - 1u); + putf(w1, 11, 11, in->nosched); + putf(w1, 10, 10, in->efo_wi0); + putf(w1, 9, 9, in->efo_wi1); + putf(w1, 22, 22, in->efo_a1lneg); + putf(w1, 23, 23, in->skipinv); + putf(w1, 26, 24, in->pred); + return 0; +} + +static int enc_src(uint32_t *w0, uint32_t *w1, unsigned slot, + const struct usse_operand *s, const char **err) +{ + unsigned code, num; + + if (chk_num(s->bank, s->num, "source", err)) + return -1; + code = src_bank_code[s->bank]; + + switch (slot) { + case 0: + if (s0_num(s, &num, err)) + return -1; + putf(w0, 20, 14, num); + putf(w1, 2, 2, s->bank == USSE_PA); + putf(w1, 7, 7, s->neg); + putf(w1, 8, 8, s->abs); + return 0; + case 1: + putf(w0, 13, 7, s->num); + putf(w0, 31, 30, code & 3); + putf(w1, 17, 17, code >> 2); + putf(w1, 5, 5, s->neg); + putf(w1, 6, 6, s->abs); + return 0; + default: + putf(w0, 6, 0, s->num); + putf(w0, 29, 28, code & 3); + putf(w1, 16, 16, code >> 2); + putf(w1, 3, 3, s->neg); + putf(w1, 4, 4, s->abs); + return 0; + } +} + +static void dec_src(uint32_t w0, uint32_t w1, unsigned slot, struct usse_operand *s) +{ + memset(s, 0, sizeof *s); + switch (slot) { + case 0: + s->num = getf(w0, 20, 14); + s->bank = getf(w1, 2, 2) ? USSE_PA : USSE_TEMP; + if (s->bank == USSE_TEMP && s->num >= 124) { + s->bank = USSE_INTERNAL; + s->num -= 124; + } + s->neg = getf(w1, 7, 7); + s->abs = getf(w1, 8, 8); + break; + case 1: + s->num = getf(w0, 13, 7); + s->bank = src_bank_from[getf(w0, 31, 30) | (getf(w1, 17, 17) << 2)]; + s->neg = getf(w1, 5, 5); + s->abs = getf(w1, 6, 6); + break; + default: + s->num = getf(w0, 6, 0); + s->bank = src_bank_from[getf(w0, 29, 28) | (getf(w1, 16, 16) << 2)]; + s->neg = getf(w1, 3, 3); + s->abs = getf(w1, 4, 4); + break; + } +} + +static int enc_dst(uint32_t *w0, uint32_t *w1, const struct usse_operand *d, + const char **err) +{ + unsigned code; + + if (d->bank >= USSE_BANK_COUNT || dst_bank_code[d->bank] == 0xFF) { + *err = "bad destination bank"; + return -1; + } + if (chk_num(d->bank, d->num, "destination", err)) + return -1; + code = dst_bank_code[d->bank]; + putf(w0, 27, 21, d->num); + putf(w1, 1, 0, code & 3); + putf(w1, 19, 19, code >> 2); + return 0; +} + +/* Iteration control: repeat mode counts, mask mode gates four iterations. */ +static int enc_rep(uint32_t *w1, const struct usse_insn *in, const char **err) +{ + if (in->repeat) { + if (in->repeat > 16) { *err = "repeat count above 16"; return -1; } + putf(w1, 21, 21, 1); + putf(w1, 15, 12, in->repeat - 1); + } else { + if (in->mask > 0xF) { *err = "write mask above 4 bits"; return -1; } + putf(w1, 15, 12, in->mask); + } + return 0; +} + +static void enc_common(uint32_t *w1, const struct usse_insn *in) +{ + putf(w1, 11, 11, in->nosched); + putf(w1, 18, 18, in->end); + putf(w1, 20, 20, in->syncstart); + putf(w1, 23, 23, in->skipinv); + putf(w1, 26, 24, in->pred); +} + +int usse_encode(const struct usse_insn *in, uint64_t *out, const char **err) +{ + uint32_t w0 = 0, w1 = 0; + unsigned grp, sub, i; + const char *dummy = ""; + + if (!err) err = &dummy; + *err = NULL; + if (in->op >= USSE_OP_COUNT) { *err = "unknown opcode"; return -1; } + + /* Refuse rather than silently drop a modifier the form cannot hold. */ + if (!usse_op_has_srcmod(in->op)) + for (i = 0; i < 3; i++) + if (in->src[i].neg || in->src[i].abs) { + *err = "this form has no source negate/absolute"; + return -1; + } + + if (alu_group(in->op, &grp, &sub) == 0) { + putf(&w1, 31, 27, grp); + putf(&w1, 10, 9, sub); + if (enc_dst(&w0, &w1, &in->dst, err) || enc_rep(&w1, in, err)) + return -1; + for (i = 0; i < 3; i++) + if ((op_srcs[in->op] >> i) & 1) + if (enc_src(&w0, &w1, i, &in->src[i], err)) + return -1; + enc_common(&w1, in); + *out = ((uint64_t)w1 << 32) | w0; + return 0; + } + + if (in->op == USSE_EFO) { + if (enc_efo(&w0, &w1, in, err)) + return -1; + *out = ((uint64_t)w1 << 32) | w0; + return 0; + } + + switch (in->op) { + case USSE_MOV: + putf(&w1, 31, 27, 0x05); + if (enc_dst(&w0, &w1, &in->dst, err) || enc_rep(&w1, in, err) || + enc_src(&w0, &w1, 1, &in->src[1], err)) + return -1; + enc_common(&w1, in); + break; + + case USSE_MOVC: + putf(&w1, 31, 27, 0x05); + putf(&w1, 10, 8, 4); /* float32 test data type */ + if (enc_dst(&w0, &w1, &in->dst, err) || enc_rep(&w1, in, err) || + enc_src(&w0, &w1, 1, &in->src[1], err) || + enc_src(&w0, &w1, 2, &in->src[2], err)) + return -1; + if (chk_num(in->src[0].bank, in->src[0].num, "movc condition", err) || + s0_num(&in->src[0], &i, err)) + return -1; + putf(&w0, 20, 14, i); + putf(&w1, 2, 2, in->src[0].bank == USSE_PA); + putf(&w1, 7, 7, in->movc_test == USSE_MOVC_TNZ); + putf(&w1, 22, 22, in->movc_test == USSE_MOVC_TN); + enc_common(&w1, in); + break; + + case USSE_UNPCKF32U8: + case USSE_PCKU8F32: + case USSE_PCKF16F32: + case USSE_UNPCKF32F16: + case USSE_UNPCKF32U16: + case USSE_UNPCKF32S16: + /* The byte mask has to cover the destination format: an f32 + * whole, an f16 as one or both halves. useasm refuses anything + * else (EncodePackInstruction, "Invalid bytemask specified"). */ + if (in->op != USSE_PCKU8F32 && in->op != USSE_PCKF16F32 && + in->bytemask != 0xF) { + *err = "an f32 unpack destination takes bytemask1111 only"; + return -1; + } + if (in->op == USSE_PCKF16F32 && in->bytemask != 0xF && + in->bytemask != 0xC && in->bytemask != 0x3) { + *err = "an f16 pack destination takes bytemask 1111, 1100 or 0011"; + return -1; + } + /* The component is a byte offset into the source register and + * the source format bounds it: any byte for the 8-bit forms, + * 0 or 2 for a 16-bit one, 0 for a 32-bit one. The vendor's + * encoder refuses the rest outright (usp/hwinst.c + * HWInstEncodePCKUNPCKInstNonVec) and the part reads the + * wrong bytes, which is a channel that comes back zero + * rather than an error. */ + { + unsigned sf = in->op == USSE_PCKU8F32 || + in->op == USSE_PCKF16F32 ? 6u : + unpck_src_fmt(in->op); + unsigned c1 = in->src[1].comp, c2 = in->src[2].comp; + + if ((sf == 3u || sf == 4u || sf == 5u) && + ((c1 != 0 && c1 != 2) || (c2 != 0 && c2 != 2))) { + *err = "a 16-bit pack source is at byte 0 or 2"; + return -1; + } + if (sf == 6u && (c1 || c2)) { + *err = "a 32-bit pack source is at byte 0"; + return -1; + } + } + putf(&w1, 31, 27, 0x08); + /* The two format fields name the destination and the source; + * unpacking is the same instruction with them swapped. + * u8 is 0, f16 is 5 and f32 is 6. */ + putf(&w1, 8, 6, in->op == USSE_PCKU8F32 ? 0 : + in->op == USSE_PCKF16F32 ? 5 : 6); + putf(&w1, 11, 9, in->op == USSE_PCKU8F32 || + in->op == USSE_PCKF16F32 ? 6 : + unpck_src_fmt(in->op)); + putf(&w1, 5, 2, in->bytemask); + putf(&w0, 18, 18, in->scale); + if (enc_dst(&w0, &w1, &in->dst, err) || enc_rep(&w1, in, err)) + return -1; + if (chk_num(in->src[1].bank, in->src[1].num, "pack source", err) || + chk_num(in->src[2].bank, in->src[2].num, "pack source", err)) + return -1; + putf(&w0, 13, 7, in->src[1].num); + putf(&w0, 31, 30, src_bank_code[in->src[1].bank] & 3); + putf(&w1, 17, 17, src_bank_code[in->src[1].bank] >> 2); + putf(&w0, 17, 16, in->src[1].comp); + putf(&w0, 6, 0, in->src[2].num); + putf(&w0, 29, 28, src_bank_code[in->src[2].bank] & 3); + putf(&w1, 16, 16, src_bank_code[in->src[2].bank] >> 2); + putf(&w0, 15, 14, in->src[2].comp); + putf(&w1, 22, 22, in->nosched); + putf(&w1, 18, 18, in->end); + putf(&w1, 20, 20, in->syncstart); + putf(&w1, 23, 23, in->skipinv); + putf(&w1, 26, 24, in->pred); + break; + + case USSE_SMP: + /* SyncStart on a sample deschedules the task for good - the + * dependent read never lands and the wait on it never + * returns. Measured; refused here so it cannot come back. */ + if (in->syncstart) { + *err = "a sample cannot carry syncstart"; + return -1; + } + /* Two of rev 1.2.1's errata, as useasm states them: a sample + * cannot be predicated (BRN25355, useasm.c:11438) and cannot + * repeat or take a mask other than .x (BRN26681, + * useasm.c:11255). */ + if (in->pred) { + *err = "a sample cannot be predicated on this core (BRN25355)"; + return -1; + } + if (in->repeat || in->mask != 1) { + *err = "a sample cannot repeat on this core (BRN26681)"; + return -1; + } + if (in->smp_drc > 1) { + *err = "the part has two dependent read counters"; + return -1; + } + putf(&w1, 31, 27, 0x1C); + if (in->smp_dim < 1 || in->smp_dim > 3) { + *err = "sample coordinate dimension must be 1, 2 or 3"; + return -1; + } + putf(&w1, 11, 10, in->smp_dim - 1); + if (in->smp_lodm > USSE_SMP_LODM_REPLACE) { + *err = "only the bias and replace LOD modes are modelled"; + return -1; + } + putf(&w1, 9, 8, in->smp_lodm); + putf(&w1, 1, 0, in->smp_drc & 3); + /* The coordinate's format. The unit reads an f16 pair out of + * one register when this is set, and two f32 registers when + * it is not. */ + putf(&w1, 3, 3, in->smp_f16 ? 1 : 0); + if (in->dst.bank != USSE_TEMP && in->dst.bank != USSE_PA) { + *err = "sample destination must be a temporary or primary attribute"; + return -1; + } + if (chk_num(in->dst.bank, in->dst.num, "sample destination", err)) + return -1; + putf(&w0, 27, 21, in->dst.num); + putf(&w1, 7, 7, in->dst.bank == USSE_PA); + if (in->src[0].bank >= USSE_BANK_COUNT || + smp_s0_code[in->src[0].bank] == 0xFF) { + *err = "sample coordinate bank not addressable by src0"; + return -1; + } + if (chk_num(in->src[0].bank, in->src[0].num, "sample coordinate", err) || + chk_num(in->src[1].bank, in->src[1].num, "sample state", err)) + return -1; + putf(&w0, 20, 14, in->src[0].num); + putf(&w1, 2, 2, smp_s0_code[in->src[0].bank] & 1); + putf(&w1, 18, 18, smp_s0_code[in->src[0].bank] >> 1); + putf(&w0, 13, 7, in->src[1].num); + putf(&w0, 31, 30, src_bank_code[in->src[1].bank] & 3); + putf(&w1, 17, 17, src_bank_code[in->src[1].bank] >> 2); + /* The LOD operand sits in the ordinary source-2 slot, as + * useasm's EncodeSMPInstruction places it (EncodeSrc2 with + * the extended bank set). Verified: "smp2dbias r0, pa0, sa4, + * r5" is 0xE0001504C0000205 and "smp2dreplace r4, pa2, sa7, + * sa9" is 0xE0001604F0808389 from useasm -target=sgx535. */ + if (in->smp_lodm) { + unsigned code; + + /* Number and bank only: the modifier bits the other + * forms keep at w1[4:3] are the coordinate type here. */ + if (chk_num(in->src[2].bank, in->src[2].num, + "sample LOD", err)) + return -1; + code = src_bank_code[in->src[2].bank]; + putf(&w0, 6, 0, in->src[2].num); + putf(&w0, 29, 28, code & 3); + putf(&w1, 16, 16, code >> 2); + } + if (enc_rep(&w1, in, err)) + return -1; + putf(&w1, 20, 20, in->syncstart); + putf(&w1, 23, 23, in->skipinv); + putf(&w1, 26, 24, in->pred); + break; + + /* TEST: an ALU operation whose result sets a predicate. The vector + * write is suppressed (w0[20] left at zero), so the destination is a + * placeholder - what the instruction produces is the predicate. + * + * The operands go in first: a source's neg and abs bits land on the + * channel-select and predicate-destination fields for this group, so + * the TEST fields have to be written over them. A modifier on a source + * therefore has nowhere to go and is refused. */ + case USSE_TEST: + if (enc_dst(&w0, &w1, &in->dst, err)) + return -1; + if (enc_src(&w0, &w1, 1, &in->src[1], err)) + return -1; + if (enc_src(&w0, &w1, 2, &in->src[2], err)) + return -1; + if (in->src[1].neg || in->src[1].abs || + in->src[2].neg || in->src[2].abs) { + *err = "test takes no source modifiers"; + return -1; + } + if (in->test_pdst > 3) { + *err = "the part has four predicates"; + return -1; + } + /* useasm.c EncodeTestInstruction_Repeat: with a repeat mask + * other than .x the predicate written has to be p0. */ + if (in->mask != 1 && in->test_pdst) { + *err = "a test with a repeat mask can only write p0"; + return -1; + } + putf(&w1, 31, 27, 0x09); + putf(&w1, 26, 24, in->pred); + putf(&w1, 23, 23, in->skipinv); + putf(&w1, 15, 12, in->mask); + putf(&w1, 11, 7, in->test_type); + putf(&w1, 6, 4, in->test_chan); + putf(&w1, 3, 2, in->test_pdst); + putf(&w0, 19, 14, in->test_alu); + putf(&w0, 20, 20, in->test_wr); + break; + + case USSE_LIMM: + putf(&w1, 31, 27, 0x1F); + putf(&w1, 26, 24, 4); + putf(&w1, 21, 20, 2); + if (enc_dst(&w0, &w1, &in->dst, err)) + return -1; + putf(&w0, 20, 0, in->imm & 0x1FFFFF); + putf(&w1, 8, 4, (in->imm >> 21) & 0x1F); + putf(&w1, 17, 12, (in->imm >> 26) & 0x3F); + putf(&w1, 11, 9, in->pred); + putf(&w1, 18, 18, in->end); + putf(&w1, 22, 22, in->nosched); + putf(&w1, 23, 23, in->skipinv); + break; + + case USSE_SMLSI: + putf(&w1, 31, 27, 0x1F); + putf(&w1, 26, 24, 2); + putf(&w1, 21, 20, 1); + for (i = 0; i < 4; i++) { + unsigned byte = in->moe_isswz[i] ? in->moe_swz[i] + : (unsigned)(in->moe_inc[i] & 0xFF); + putf(&w0, 31 - 8 * i, 24 - 8 * i, byte); + putf(&w1, 3 - i, 3 - i, in->moe_isswz[i] != 0); + } + break; + + case USSE_EMIT: + putf(&w1, 31, 27, 0x1F); + putf(&w1, 26, 24, 3); + putf(&w1, 21, 20, 2); + putf(&w1, 17, 16, 3); /* constant in every captured emit */ + putf(&w1, 14, 12, in->emit_target); + putf(&w1, 18, 18, in->end); + putf(&w0, 31, 31, 1); + putf(&w0, 29, 29, 1); + putf(&w0, 21, 21, in->emit_freep); + break; + + /* Wait for the dependent read the sampler started. Without it the + * program reads the texel register before the texture unit has + * written it. */ + case USSE_WDF: + if (in->smp_drc > 1) { + *err = "the part has two dependent read counters"; + return -1; + } + putf(&w1, 31, 27, 0x1F); + putf(&w1, 26, 24, 1); + putf(&w1, 21, 20, 2); + putf(&w1, 1, 0, in->smp_drc & 3); + break; + + case USSE_NOP: /* word1 bit 23 is SyncStart here, not the end flag */ + putf(&w1, 31, 27, 0x1F); + putf(&w1, 8, 6, 5); + putf(&w1, 23, 23, in->syncstart); + putf(&w1, 18, 18, in->end); + break; + + case USSE_BA: + case USSE_BR: + putf(&w1, 31, 27, 0x1F); + putf(&w1, 8, 6, in->op == USSE_BA ? 0 : 1); + putf(&w1, 26, 24, in->pred); + if (in->offset < -2048 || in->offset > 2047) { + *err = "branch offset outside the 12-bit signed range"; + return -1; + } + putf(&w0, 11, 0, (uint32_t)in->offset & 0xFFF); + break; + + default: + *err = "opcode has no encoder"; + return -1; + } + + *out = ((uint64_t)w1 << 32) | w0; + return 0; +} + +static void dec_common(uint32_t w1, struct usse_insn *o) +{ + o->nosched = getf(w1, 11, 11); + o->end = getf(w1, 18, 18); + o->syncstart = getf(w1, 20, 20); + o->skipinv = getf(w1, 23, 23); + o->pred = getf(w1, 26, 24); + if (getf(w1, 21, 21)) + o->repeat = getf(w1, 15, 12) + 1; + else + o->mask = getf(w1, 15, 12); +} + +int usse_decode(uint64_t word, struct usse_insn *o) +{ + uint32_t w0 = (uint32_t)word, w1 = (uint32_t)(word >> 32); + unsigned grp = GRP(w1), sub = getf(w1, 10, 9), i; + + memset(o, 0, sizeof *o); + + switch (grp) { + case 0x00: o->op = USSE_MAD + sub; break; + case 0x01: o->op = USSE_RCP + sub; break; + case 0x02: if (sub) return -1; o->op = USSE_DP; break; + case 0x03: if (sub > 1) return -1; o->op = USSE_MIN + sub; break; + case 0x04: if (sub > 1) return -1; o->op = USSE_DSX + sub; break; + case 0x05: + if (getf(w1, 10, 8) == 0) + o->op = USSE_MOV; + else if (getf(w1, 10, 8) == 4) + o->op = USSE_MOVC; + else + return -1; + break; + case 0x07: { + static const uint8_t s12_from[4] = { USSE_TEMP, USSE_OUT, + USSE_PA, USSE_SA }; + static const uint8_t d_from[4] = { USSE_TEMP, USSE_OUT, + USSE_PA, USSE_INTERNAL }; + struct usse_operand *s; + + o->op = USSE_EFO; + o->efo_dsrc = (uint8_t)getf(w1, 21, 20); + o->efo_isrc = (uint8_t)getf(w1, 19, 18); + o->efo_asrc = (uint8_t)getf(w1, 17, 16); + o->efo_msrc = (uint8_t)getf(w1, 15, 14); + o->repeat = (uint8_t)(getf(w1, 13, 12) + 1u); + o->nosched = (uint8_t)getf(w1, 11, 11); + o->efo_wi0 = (uint8_t)getf(w1, 10, 10); + o->efo_wi1 = (uint8_t)getf(w1, 9, 9); + o->efo_a1lneg = (uint8_t)getf(w1, 22, 22); + o->skipinv = (uint8_t)getf(w1, 23, 23); + o->pred = (uint8_t)getf(w1, 26, 24); + o->dst.bank = d_from[getf(w1, 1, 0)]; + o->dst.num = (uint8_t)getf(w0, 27, 21); + o->src[0].bank = getf(w1, 2, 2) ? USSE_PA : USSE_TEMP; + o->src[0].num = (uint8_t)getf(w0, 20, 14); + o->src[1].bank = s12_from[getf(w0, 31, 30)]; + o->src[1].num = (uint8_t)getf(w0, 13, 7); + o->src[2].bank = s12_from[getf(w0, 29, 28)]; + o->src[2].num = (uint8_t)getf(w0, 6, 0); + for (i = 0; i < 3; i++) { + unsigned mod = getf(w1, 8 - 2 * i, 7 - 2 * i); + + o->src[i].neg = (uint8_t)(mod & 1); + o->src[i].abs = (uint8_t)((mod >> 1) & 1); + } + /* The top four of the temporary bank are i0..i3 in every slot. + * The destination's fourth bank code is the dedicated internal + * bank, which the vendor's own output does not use. */ + for (i = 0; i < 4; i++) { + s = i ? &o->src[i - 1] : &o->dst; + if (s->bank == USSE_TEMP && s->num >= 124) { + s->bank = USSE_INTERNAL; + s->num = (uint8_t)(s->num - 124u); + } + } + return 0; + } + case 0x08: { + unsigned df = getf(w1, 8, 6), sf = getf(w1, 11, 9); + + if (df == 0 && sf == 6) + o->op = USSE_PCKU8F32; + else if (df == 6 && sf == 0) + o->op = USSE_UNPCKF32U8; + else if (df == 5 && sf == 6) + o->op = USSE_PCKF16F32; + else if (df == 6 && sf == 5) + o->op = USSE_UNPCKF32F16; + else if (df == 6 && sf == 3) + o->op = USSE_UNPCKF32U16; + else if (df == 6 && sf == 4) + o->op = USSE_UNPCKF32S16; + else + return -1; + break; + } + case 0x09: + o->op = USSE_TEST; + o->test_type = (uint8_t)getf(w1, 11, 7); + o->test_chan = (uint8_t)getf(w1, 6, 4); + o->test_pdst = (uint8_t)getf(w1, 3, 2); + o->test_alu = (uint8_t)getf(w0, 19, 14); + o->test_wr = (uint8_t)getf(w0, 20, 20); + o->mask = (uint8_t)getf(w1, 15, 12); + o->skipinv = (uint8_t)getf(w1, 23, 23); + o->pred = (uint8_t)getf(w1, 26, 24); + dec_src(w0, w1, 1, &o->src[1]); + dec_src(w0, w1, 2, &o->src[2]); + o->src[1].neg = o->src[1].abs = 0; + o->src[2].neg = o->src[2].abs = 0; + o->dst.bank = dst_bank_from[getf(w1, 1, 0) | + (getf(w1, 19, 19) << 2)]; + o->dst.num = (uint8_t)getf(w0, 27, 21); + return 0; + case 0x1C: + if (getf(w1, 9, 8) == 3 || getf(w1, 11, 10) == 3) + return -1; + o->op = USSE_SMP; + break; + case 0x1F: + if (getf(w1, 21, 20) == 0 && getf(w1, 8, 6) == 0) o->op = USSE_BA; + else if (getf(w1, 21, 20) == 0 && getf(w1, 8, 6) == 1) o->op = USSE_BR; + else if (getf(w1, 21, 20) == 0 && getf(w1, 8, 6) == 5) o->op = USSE_NOP; + else if (getf(w1, 21, 20) == 1 && getf(w1, 26, 24) == 2) o->op = USSE_SMLSI; + else if (getf(w1, 21, 20) == 2 && getf(w1, 26, 24) == 4) o->op = USSE_LIMM; + else if (getf(w1, 21, 20) == 2 && getf(w1, 26, 24) == 3) o->op = USSE_EMIT; + else if (getf(w1, 21, 20) == 2 && getf(w1, 26, 24) == 1) { + o->op = USSE_WDF; + o->smp_drc = (uint8_t)getf(w1, 1, 0); + } else return -1; + break; + default: + return -1; + } + + switch (o->op) { + case USSE_LIMM: + o->dst.num = getf(w0, 27, 21); + o->dst.bank = dst_bank_from[getf(w1, 1, 0) | (getf(w1, 19, 19) << 2)]; + o->imm = getf(w0, 20, 0) | (getf(w1, 8, 4) << 21) | (getf(w1, 17, 12) << 26); + o->pred = getf(w1, 11, 9); + o->end = getf(w1, 18, 18); + o->nosched = getf(w1, 22, 22); + o->skipinv = getf(w1, 23, 23); + return 0; + case USSE_SMLSI: + for (i = 0; i < 4; i++) { + unsigned byte = getf(w0, 31 - 8 * i, 24 - 8 * i); + o->moe_isswz[i] = getf(w1, 3 - i, 3 - i); + if (o->moe_isswz[i]) + o->moe_swz[i] = (uint8_t)byte; + else + o->moe_inc[i] = (int8_t)byte; + } + return 0; + case USSE_EMIT: + o->emit_target = getf(w1, 14, 12); + o->end = getf(w1, 18, 18); + o->emit_freep = getf(w0, 21, 21); + return 0; + case USSE_NOP: + o->syncstart = getf(w1, 23, 23); + return 0; + case USSE_BA: + case USSE_BR: + o->pred = getf(w1, 26, 24); + o->offset = (int32_t)(getf(w0, 11, 0) << 20) >> 20; + return 0; + case USSE_SMP: + o->smp_dim = getf(w1, 11, 10) + 1; + o->smp_drc = getf(w1, 1, 0); + o->dst.num = getf(w0, 27, 21); + o->dst.bank = getf(w1, 7, 7) ? USSE_PA : USSE_TEMP; + o->src[0].num = getf(w0, 20, 14); + o->src[0].bank = smp_s0_from[getf(w1, 2, 2) | (getf(w1, 18, 18) << 1)]; + o->src[1].num = getf(w0, 13, 7); + o->src[1].bank = src_bank_from[getf(w0, 31, 30) | (getf(w1, 17, 17) << 2)]; + o->syncstart = getf(w1, 20, 20); + o->skipinv = getf(w1, 23, 23); + o->pred = getf(w1, 26, 24); + if (getf(w1, 21, 21)) o->repeat = getf(w1, 15, 12) + 1; + else o->mask = getf(w1, 15, 12); + o->smp_f16 = getf(w1, 3, 3); + o->smp_lodm = (uint8_t)getf(w1, 9, 8); + if (o->smp_lodm) { + dec_src(w0, w1, 2, &o->src[2]); + o->src[2].neg = o->src[2].abs = 0; + } + return 0; + case USSE_PCKU8F32: + case USSE_UNPCKF32U8: + case USSE_PCKF16F32: + case USSE_UNPCKF32F16: + case USSE_UNPCKF32U16: + case USSE_UNPCKF32S16: + o->dst.num = getf(w0, 27, 21); + o->dst.bank = dst_bank_from[getf(w1, 1, 0) | (getf(w1, 19, 19) << 2)]; + o->bytemask = getf(w1, 5, 2); + o->scale = getf(w0, 18, 18); + o->src[1].num = getf(w0, 13, 7); + o->src[1].bank = src_bank_from[getf(w0, 31, 30) | (getf(w1, 17, 17) << 2)]; + o->src[1].comp = getf(w0, 17, 16); + o->src[2].num = getf(w0, 6, 0); + o->src[2].bank = src_bank_from[getf(w0, 29, 28) | (getf(w1, 16, 16) << 2)]; + o->src[2].comp = getf(w0, 15, 14); + o->end = getf(w1, 18, 18); + o->syncstart = getf(w1, 20, 20); + o->nosched = getf(w1, 22, 22); + o->skipinv = getf(w1, 23, 23); + o->pred = getf(w1, 26, 24); + if (getf(w1, 21, 21)) o->repeat = getf(w1, 15, 12) + 1; + else o->mask = getf(w1, 15, 12); + return 0; + default: + break; + } + + o->dst.num = getf(w0, 27, 21); + o->dst.bank = dst_bank_from[getf(w1, 1, 0) | (getf(w1, 19, 19) << 2)]; + for (i = 0; i < 3; i++) + if ((op_srcs[o->op] >> i) & 1) + dec_src(w0, w1, i, &o->src[i]); + dec_common(w1, o); + if (o->op == USSE_MOVC) { + dec_src(w0, w1, 0, &o->src[0]); + o->src[0].neg = o->src[0].abs = 0; + o->movc_test = getf(w1, 22, 22) ? USSE_MOVC_TN + : (getf(w1, 7, 7) ? USSE_MOVC_TNZ : USSE_MOVC_TZ); + } + return 0; +} + +/* --- text, in Imagination's disassembly syntax ------------------------- */ + +static char *put(char *p, char *end, const char *fmt, ...) +{ + va_list ap; + int n; + + if (p >= end) + return p; + va_start(ap, fmt); + n = vsnprintf(p, (size_t)(end - p), fmt, ap); + va_end(ap); + return n < 0 ? p : (p + n > end ? end : p + n); +} + +/* The vendor renders the upper half of the constant bank as the global bank. */ +static char *put_reg(char *p, char *end, unsigned bank, unsigned num) +{ + if (bank == USSE_INDEX) + return put(p, end, "r[il + #%u]", num); + if (bank == USSE_CONST && num >= 64) + return put(p, end, "g%u", num - 64); + return put(p, end, "%s%u", bank_pfx[bank], num); +} + +static char *put_src(char *p, char *end, const struct usse_operand *s, int pck) +{ + p = put_reg(p, end, s->bank, s->num); + if (pck) + p = put(p, end, ".%u", s->comp); + if (s->neg) + p = put(p, end, ".neg"); + if (s->abs) + p = put(p, end, ".abs"); + return p; +} + +static char *put_dst(char *p, char *end, const struct usse_insn *in) +{ + static const char ch[4] = { 'x', 'y', 'z', 'w' }; + unsigned i; + + p = put_reg(p, end, in->dst.bank, in->dst.num); + /* The vendor leaves a full byte mask unsaid. */ + if (is_pck(in->op) && in->bytemask != 0xF) + p = put(p, end, ".bytemask%u%u%u%u", (in->bytemask >> 3) & 1, + (in->bytemask >> 2) & 1, (in->bytemask >> 1) & 1, + in->bytemask & 1); + if ((op_feat(in->op) & F_MASK) && !in->repeat && in->mask != 1) { + p = put(p, end, "."); + for (i = 0; i < 4; i++) + if (in->mask & (1u << i)) + p = put(p, end, "%c", ch[i]); + } + return p; +} + +void usse_text(const struct usse_insn *in, char *buf, size_t n) +{ + static const char *const pred_pfx[8] = { + "", "p0 ", "p1 ", "p2 ", "p3 ", "!p0 ", "!p1 ", "Pn " + }; + static const char *const movc_sfx[3] = { ".tz", ".tnz", ".tn" }; + char *p = buf, *end = buf + n; + unsigned i, feat; + + if (!n) + return; + buf[0] = 0; + feat = op_feat(in->op); + + if (in->op == USSE_SMLSI) { + p = put(p, end, "smlsi "); + for (i = 0; i < 4; i++) { + if (in->moe_isswz[i]) { + static const char c[4] = { 'x', 'y', 'z', 'w' }; + p = put(p, end, "swizzle(%c%c%c%c)", + c[in->moe_swz[i] & 3], c[(in->moe_swz[i] >> 2) & 3], + c[(in->moe_swz[i] >> 4) & 3], c[(in->moe_swz[i] >> 6) & 3]); + } else { + p = put(p, end, "#%d", in->moe_inc[i]); + } + p = put(p, end, ", "); + } + for (i = 0; i < 4; i++) + p = put(p, end, "%s, ", in->moe_isswz[i] ? "swizzlemode" + : "incrementmode"); + put(p, end, "#0, #0, #0"); + return; + } + + /* The vendor's own rendering, which the oracle compares against: + * "efo.skipinv o1= i1, i0 = m0, i1 = m1, a0=m0+m1, a1=i1+i0, + * m0=src0*src1, m1=src0*src2, pa0, sa12, sa13". The adder line for + * ASRC 3 carries no comma between its halves; that is the + * disassembler's own spelling, not a slip. */ + if (in->op == USSE_EFO) { + static const char *const dsel[4] = { "i0", "i1", "a0", "a1" }; + static const char *const isel[4][2] = { + { "a0", "a1" }, { "a1", "a0" }, + { "m0", "m1" }, { "a0", "m1" } + }; + static const char *const add[4][2] = { + { "a0=m0+m1", "a1=%si1+i0" }, + { "a0=m0+src2", "a1=%si1+i0" }, + { "a0=i0+m0", "a1=%si1+m1" }, + { "a0=src0+src1", "a1=%ssrc2+src0" } + }; + static const char *const mul[4] = { + "m0=src0*src1, m1=src0*src2", + "m0=src0*src1, m1=src0*src0", + "m0=src1*src2, m1=src0*src0", + "m0=src1*i0, m1=src0*i1" + }; + const char *neg = in->efo_a1lneg ? "-" : ""; + + p = put(p, end, "%sefo", pred_pfx[in->pred & 7]); + if (in->skipinv) p = put(p, end, ".skipinv"); + if (in->nosched) p = put(p, end, ".nosched"); + if (in->repeat > 1) p = put(p, end, ".repeat%u", in->repeat); + p = put(p, end, " "); + p = put_reg(p, end, in->dst.bank, in->dst.num); + p = put(p, end, "= %s, ", dsel[in->efo_dsrc & 3]); + p = put(p, end, "%si0 = %s, ", in->efo_wi0 ? "" : "!", + isel[in->efo_isrc & 3][0]); + p = put(p, end, "%si1 = %s, ", in->efo_wi1 ? "" : "!", + isel[in->efo_isrc & 3][1]); + p = put(p, end, "%s%s", add[in->efo_asrc & 3][0], + (in->efo_asrc & 3) == 3 ? " " : ", "); + p = put(p, end, add[in->efo_asrc & 3][1], neg); + p = put(p, end, ", %s", mul[in->efo_msrc & 3]); + for (i = 0; i < 3; i++) { + p = put(p, end, ", "); + p = put_src(p, end, &in->src[i], 0); + } + return; + } + + if (in->op == USSE_EMIT) { + static const char *const tgt[8] = { + "emitpix2", "emitpix2", "emitpix2", "emitpix2", + "emitst", "emitvtx", "emitprimitive", "?" + }; + p = put(p, end, "%s", tgt[in->emit_target & 7]); + if (in->end) p = put(p, end, ".end"); + if (in->emit_freep) p = put(p, end, ".freep"); + /* The pixel form carries extra operands whose fields are not + * decoded; every encoding we can produce leaves them zero. */ + if (in->emit_target < 4) + put(p, end, " #0 /* incp */, r0, #0, #0, #0x00000000"); + else + put(p, end, " #0 /* incp */, #0"); + return; + } + + p = put(p, end, "%s", (feat & F_PRED) ? pred_pfx[in->pred & 7] : ""); + + if (in->op == USSE_SMP) + p = put(p, end, "smp%ud%s", in->smp_dim, + in->smp_lodm == USSE_SMP_LODM_BIAS ? "bias" : + in->smp_lodm == USSE_SMP_LODM_REPLACE ? "replace" : ""); + else if (in->op == USSE_TEST && (in->test_alu >> 4) == 3) + /* The bitwise unit's ops, as the vendor names the test by. */ + p = put(p, end, "%s", (in->test_alu & 0xF) < 8 ? + (const char *const[]){ "and", "or", "xor", "shl", "shr", + "rol", "?", "asr" }[in->test_alu & 0xF] : + "?"); + else + p = put(p, end, "%s", op_names[in->op]); + + if (in->skipinv && usse_op_has_skipinv(in->op)) p = put(p, end, ".skipinv"); + if (in->syncstart && (feat & F_SYNCS)) p = put(p, end, ".syncs"); + if (in->nosched && (feat & F_NOSCH)) p = put(p, end, ".nosched"); + if (in->end && (feat & F_END)) p = put(p, end, ".end"); + if (in->op == USSE_MOVC) p = put(p, end, ".flt"); + if (is_pck(in->op) && in->scale) p = put(p, end, ".scale"); + if (in->repeat > 1 && op_has_repeat(in->op)) + p = put(p, end, ".repeat%u", in->repeat); + if (in->op == USSE_MOVC) p = put(p, end, "%s", movc_sfx[in->movc_test % 3]); + /* test, the vendor's own spelling: sign t/n/p + * from w1[11:10], combine o/a from w1[7], zero t/z/nz from w1[9:8]. */ + if (in->op == USSE_TEST) { + static const char sign[4] = { 't', 'n', 'p', '?' }; + static const char *const zero[4] = { "t", "z", "nz", "?" }; + + p = put(p, end, ".test%c%c%s.chan%u", + sign[(in->test_type >> 3) & 3], + (in->test_type & 1) ? 'a' : 'o', + zero[(in->test_type >> 1) & 3], in->test_chan); + } + + if (in->op == USSE_WDF) { + put(p, end, " drc%u", in->smp_drc & 3); + return; + } + if (in->op == USSE_NOP) + return; + + if (in->op == USSE_BA || in->op == USSE_BR) { + if (in->offset < 0) + put(p, end, " -#0x%08X", (unsigned)-in->offset); + else + put(p, end, " #0x%08X", (unsigned)in->offset); + return; + } + + p = put(p, end, " "); + /* The vector write is suppressed on a test, which the disassembly + * marks by negating the destination; the predicate it does write + * follows it. */ + if (in->op == USSE_TEST && !in->test_wr) + p = put(p, end, "!"); + p = put_dst(p, end, in); + if (in->op == USSE_TEST) + p = put(p, end, ", p%u", in->test_pdst); + + if (in->op == USSE_LIMM) { + put(p, end, ", #0x%08X", in->imm); + return; + } + if (in->op == USSE_SMP) { + p = put(p, end, ", "); + p = put_src(p, end, &in->src[0], 0); + if (in->smp_f16) + p = put(p, end, ".flt16"); + p = put(p, end, ", "); + p = put_src(p, end, &in->src[1], 0); + if (in->smp_lodm) { + p = put(p, end, ", "); + p = put_src(p, end, &in->src[2], 0); + } + put(p, end, ", drc%u", in->smp_drc); + return; + } + + for (i = 0; i < 3; i++) { + if (!((op_srcs[in->op] >> i) & 1)) + continue; + p = put(p, end, ", "); + p = put_src(p, end, &in->src[i], is_pck(in->op)); + } + if (is_pck(in->op)) + put(p, end, ", nearest"); +} diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/usse_isa.h mesa-26.2.2/src/gallium/drivers/sgx/usse_isa.h --- mesa-26.2.2.orig/src/gallium/drivers/sgx/usse_isa.h 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/usse_isa.h 2026-09-08 10:57:36.683537691 +0200 @@ -0,0 +1,203 @@ +/* + * usse_isa.h - SGX535 USSE instruction encoder/decoder + * + * Field layout recovered by exhaustive single-bit differential disassembly + * against Imagination's own USSE disassembler at SGX535 rev 1.2.1; see + * backend/README.md for the derivation and the residual unknowns. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#ifndef USSE_ISA_H +#define USSE_ISA_H + +#include +#include + +/* Logical register banks. The hardware numbers them differently for the + * destination and for sources (slots 3 and 4 are swapped), so callers use + * this enum and the encoder maps it. */ +enum usse_bank { + USSE_TEMP = 0, + USSE_OUT, + USSE_PA, + USSE_SA, + USSE_CONST, + USSE_IMMB, + USSE_INTERNAL, + USSE_INDEX, + USSE_BANK_COUNT +}; + +/* Hardwired float constants, confirmed from vendor codegen: c48..c51 read + * 0.0 and c52..c55 read 1.0. See README "hardware constants". */ +#define USSE_C_ZERO 48 +#define USSE_C_ONE 52 + +/* The upper half of the special bank is the global registers (useasm.c + * CheckAndEncodeGlobalRegisterNumber: g encodes as the SPECIAL bank with + * EURASIA_USE_SPECIAL_INTERNALDATA, 0x40, set). g16 is BFCONTROL, whose bit 0 + * the ISP sets by the triangle's winding - measured set for GL's front face + * under GL_CCW with no flip, see sgx_face_swap_of(). */ +#define USSE_G_BASE 0x40 +#define USSE_G_BFCONTROL (USSE_G_BASE + 16) + +enum usse_op { + USSE_MAD = 0, /* dst = s0 * s1 + s2 */ + USSE_ADM, /* semantics not determined */ + USSE_MSA, /* semantics not determined */ + USSE_SUBFLR, /* dst = s1 - floor(s2) */ + USSE_RCP, USSE_RSQ, USSE_LOG, USSE_EXP, + USSE_DP, /* iteration count = term count */ + USSE_MIN, USSE_MAX, + USSE_DSX, USSE_DSY, /* screen-space derivatives, group 0x04 */ + USSE_MOV, + USSE_MOVC, /* dst = test(s0) ? s1 : s2 */ + USSE_PCKU8F32, + USSE_UNPCKF32U8, /* one byte of a packed dword to a float */ + USSE_SMP, + USSE_LIMM, + USSE_SMLSI, + USSE_EMIT, + USSE_NOP, + USSE_BA, USSE_BR, + USSE_TEST, /* set a predicate from an ALU result */ + USSE_WDF, /* wait for a dependent read to land */ + USSE_PCKF16F32, /* two floats to an f16 pair in one register */ + /* One half, u16 or s16 of a register to a float; the source's comp + * selects the half. The same group as the byte unpack with the + * source-format field set to F16 (5), U16 (3) or S16 (4) - the + * DDK's EURASIA_USE1_PCK_FMT_* (sgxdefs.h:5455-5462). */ + USSE_UNPCKF32F16, + USSE_UNPCKF32U16, + USSE_UNPCKF32S16, + /* The extended float operation: two multipliers and two adders in one + * instruction, one result to the unified store and the other two to + * the internal registers i0 and i1. sgxdefs.h:5141 EURASIA_USE1_OP_EFO + * and the EURASIA_USE1_EFO_* fields at 5398-5427. */ + USSE_EFO, + USSE_OP_COUNT +}; + +/* EFO wiring. The four selects name what feeds the two multipliers, the two + * adders, the two internal registers and the unified-store destination; the + * names are Imagination's own (sgxdefs.h:5402-5427). */ +/* Which result the unified-store destination takes, w1[21:20]. */ +enum usse_efo_dsrc { USSE_EFO_D_I0 = 0, USSE_EFO_D_I1, USSE_EFO_D_A0, + USSE_EFO_D_A1 }; +/* What i0 and i1 take, w1[19:18]. */ +enum usse_efo_isrc { USSE_EFO_I_A0A1 = 0, USSE_EFO_I_A1A0, USSE_EFO_I_M0M1, + USSE_EFO_I_A0M1 }; +/* The adders, w1[17:16]. A1LNEG negates the left-hand input of a1. */ +enum usse_efo_asrc { USSE_EFO_A_M0M1_I1I0 = 0, USSE_EFO_A_M0S2_I1I0, + USSE_EFO_A_M0I0_I1M1, USSE_EFO_A_S0S1_S2S0 }; +/* The multipliers, w1[15:14]. */ +enum usse_efo_msrc { USSE_EFO_M_S0S1_S0S2 = 0, USSE_EFO_M_S0S1_S0S0, + USSE_EFO_M_S1S2_S0S0, USSE_EFO_M_S1I0_S0I1 }; + +/* Iterations one EFO can run, w1[13:12] holding count - 1 + * (sgxdefs.h:5425 EURASIA_USE1_EFO_RCOUNT_MAX). */ +#define USSE_EFO_MAX_REPEAT 4 + +/* TEST condition, w1[11:7]: the sign term in [11:10] and the zero term in + * [9:7]. "poz" is positive or zero, which is the survive condition for a + * discard written as "kill where the value is negative". */ +#define USSE_TEST_POZ 18u + +/* TEST ALU sub-operation, w0[19:14]. */ +#define USSE_TEST_FADD 0u + +/* MOVC test on src0, float form. */ +enum usse_movc_test { USSE_MOVC_TZ = 0, USSE_MOVC_TNZ, USSE_MOVC_TN }; + +enum usse_emit_target { USSE_EMIT_PIX2 = 0, USSE_EMIT_STATE = 4, + USSE_EMIT_VTX = 5, USSE_EMIT_PRIM = 6 }; + +struct usse_operand { + uint8_t bank; + uint8_t num; + uint8_t neg; + uint8_t abs; + uint8_t comp; /* PCK only: which f32 of the source pair */ +}; + +struct usse_insn { + uint8_t op; + struct usse_operand dst; + struct usse_operand src[3]; + + uint8_t mask; /* iteration mask, used when repeat == 0 */ + uint8_t repeat; /* 1..16 iterations; 0 selects mask mode */ + uint8_t skipinv, nosched, end, syncstart; + uint8_t pred; /* 0 none, 1..4 = p0..p3, 5..6 = !p0,!p1 */ + + uint8_t movc_test; + uint8_t test_type; /* TEST: condition, w1[11:7] */ + /* The condition is named test by Imagination's + * own disassembler: sign is t/n/p (always, negative, positive), + * combine is o/a (or, and), zero is t/z/nz. So "and" of an always-true + * sign test with a zero test is a plain comparison against zero. */ +#define USSE_TEST_TAZ 0x03 /* result == 0 (testtaz) */ +#define USSE_TEST_TANZ 0x05 /* result != 0 (testtanz) */ +#define USSE_TEST_NAT 0x09 /* result < 0 (testnat) */ +#define USSE_TEST_ALU_FADD 0x00 + /* The ALU select in w0[19:18] (sgxdefs.h EURASIA_USE0_TEST_ALUSEL_*) + * above the op in w0[17:14]: 3 is the bitwise unit and its AND is 0, + * which is the form the vendor's compiler reads the face bit with + * (usc2/icvt_core.c CheckFaceType: and.testnz on g16 and #1). */ +#define USSE_TEST_ALU_AND 0x30 + uint8_t test_chan; /* TEST: channel select, w1[6:4] */ + uint8_t test_alu; /* TEST: ALU sub-operation, w0[19:14] */ + uint8_t test_pdst; /* TEST: predicate written, p0..p3 */ + /* TEST: the ALU result is written as well as the predicate, w0[20]. + * The vendor's assembler refuses the suppressed form for a bitwise + * ALU, so that form always carries it. */ + uint8_t test_wr; + uint8_t smp_dim; /* 1,2,3 */ + uint8_t smp_drc; + uint8_t smp_f16; /* coordinate is an f16 pair, w1[3] */ + /* The LOD mode, w1[9:8]: EURASIA_USE1_SMP_LODM_* (sgxdefs.h:6339). + * BIAS and REPLACE take the value from src[2], in the coordinate's + * format. GRADIENTS is not modelled. */ +#define USSE_SMP_LODM_NONE 0 +#define USSE_SMP_LODM_BIAS 1 +#define USSE_SMP_LODM_REPLACE 2 + uint8_t smp_lodm; + uint8_t bytemask; /* PCK destination byte mask */ + uint8_t scale; /* PCK: scale f32 [0,1] to the byte range */ + uint8_t emit_target; + uint8_t emit_freep; + + /* EFO: the four selects, the two internal-register write enables and + * the a1 left-input negate. */ + uint8_t efo_dsrc, efo_isrc, efo_asrc, efo_msrc; + uint8_t efo_wi0, efo_wi1; + uint8_t efo_a1lneg; + + uint32_t imm; /* LIMM */ + int32_t offset; /* BA/BR, in instructions */ + + int8_t moe_inc[4]; /* SMLSI: dst, s0, s1, s2 */ + uint8_t moe_swz[4]; + uint8_t moe_isswz[4]; +}; + +/* Encode one instruction. Returns 0, or -1 with *err set to a static string + * when the request cannot be represented. */ +int usse_encode(const struct usse_insn *in, uint64_t *out, const char **err); + +/* Decode. Returns 0 when the word is one of the forms modelled here, -1 when + * the opcode is outside that subset (not an error, just unmodelled). */ +int usse_decode(uint64_t word, struct usse_insn *out); +int usse_op_has_dst(unsigned op); + +/* Render in Imagination's disassembly syntax, for oracle comparison. */ +void usse_text(const struct usse_insn *in, char *buf, size_t n); + +const char *usse_op_name(unsigned op); +int usse_op_has_skipinv(unsigned op); +unsigned usse_op_srcs(unsigned op); /* bitmask of source slots used */ +int usse_op_has_srcmod(unsigned op); /* form carries negate/absolute bits */ + +#endif /* USSE_ISA_H */ diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/xpsb_3d.c mesa-26.2.2/src/gallium/drivers/sgx/xpsb_3d.c --- mesa-26.2.2.orig/src/gallium/drivers/sgx/xpsb_3d.c 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/xpsb_3d.c 2026-09-08 10:57:36.684606062 +0200 @@ -0,0 +1,511 @@ +/* Open replacement for Xpsb.so - choosing between the SGX and the CPU path. + * + * Two rules shape everything here. A command stream that is only partly + * correct still submits and still wedges the hardware, so an operation the + * frame layer cannot express is declined rather than approximated; and the + * decision is made before the frame is touched, so a decline never leaves a + * half-configured frame behind or a silently wrong render in front. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#include +#include +#include + +#include "xpsb_3d.h" + +#include "xpsb_frame.h" +#include "xpsb_pixel.h" +#include "xpsb_shader.h" +#include "xpsb_vidshader.h" +#include "xpsb_yuv.h" + +/* What the frame layer can actually submit today, and why the other bits are + * clear. Each is a limit stated in xpsb_frame.c, not a guess: + * + * MASK_TEX xpsb_frame_set_ntex() now builds the whole two-unit stream and + * xpsb_frame_submit() no longer refuses it, but three of its words + * are extrapolated from captures rather than taken from one: the + * two iterator control words' coordinate-set nibble, the vertex + * USE program's emit count, and the user draw record's granule + * count. A stream that is wrong in any of them still submits, so + * the bit stays clear until a run on hardware says otherwise - + * build with -DXPSB_3D_FRAME_CAPS to make that run + * + * VIDEO is open for the two packed FOURCCs only; see frame_set_vidshader() + * below and section 7 of disasm/video-integration.md for what the planar ones + * would still need. SCALAR is open. A scalar operand is one constant in sa0, loaded by the + * secondary PDS program xpsb_frame_set_scalar() builds; the sampled unit stays + * at one, because xpsb_composite_shader() reads pa0 for whichever operand is + * not the constant. That covers a source texture with a scalar mask and a + * scalar source with a mask texture - between them every composite except + * source and mask both textures, which is what MASK_TEX still gates. + * Raising a bit whose frame-side step is missing re-declines at that step, + * which is a fallback rather than a wrong render. */ +#ifndef XPSB_3D_FRAME_CAPS +#define XPSB_3D_FRAME_CAPS (XPSB_3D_CAP_QUAD | XPSB_3D_CAP_COMPOSITE | \ + XPSB_3D_CAP_SCALAR | XPSB_3D_CAP_VIDEO) +#endif + +unsigned xpsb_3d_caps(void) +{ + const char *e = getenv("XPSB_NO_3D"); + + if (e && *e && *e != '0') + return 0; + return XPSB_3D_FRAME_CAPS; +} + +/* ---- pure predicates ---- */ + +int xpsb_3d_format(uint32_t pict_format, uint32_t *fmt) +{ + struct xpsb_format f; + uint32_t code; + + if (xpsb_format_parse(pict_format, &f)) + return -1; + + if (f.bpp == 8 && f.bits[0] == 8 && !f.bits[1] && !f.bits[2] && + !f.bits[3]) + code = XPSB_FMT_A8; + else if (f.bpp == 16 && f.bits[0] == 0 && f.bits[1] == 5 && + f.bits[2] == 6 && f.bits[3] == 5 && f.shift[1] == 11 && + f.shift[3] == 0) + code = XPSB_FMT_565; + else if (f.bpp == 16 && f.bits[0] <= 1 && f.bits[1] == 5 && + f.bits[2] == 5 && f.bits[3] == 5 && f.shift[1] == 10 && + f.shift[3] == 0) + code = XPSB_FMT_1555; + else if (f.bpp == 32 && (f.bits[0] == 0 || f.bits[0] == 8) && + f.bits[1] == 8 && f.bits[2] == 8 && f.bits[3] == 8 && + f.shift[1] == 16 && f.shift[3] == 0) + code = XPSB_FMT_8888; + else if (f.bpp == 32 && (f.bits[0] == 0 || f.bits[0] == 8) && + f.bits[1] == 8 && f.bits[2] == 8 && f.bits[3] == 8 && + f.shift[1] == 0 && f.shift[3] == 16) + code = XPSB_FMT_BGR8888; + else + return -1; + + if (fmt) + *fmt = code; + return 0; +} + +static void to_desc(struct xpsb_surface_desc *d, const struct xpsb_3d_surf *s, + uint32_t fmt) +{ + memset(d, 0, sizeof *d); + d->handle = s->handle; + d->offset = s->offset; + d->w = s->w; + d->h = s->h; + d->stride = s->stride; + d->format = fmt; + d->umode = s->umode; + d->vmode = s->vmode; + d->minfilter = s->minfilter; + d->magfilter = s->magfilter; +} + +/* A destination is also the render target, so on top of the stride rule it + * has to be a colour format the raster pass can write and large enough for + * xpsb_frame_set_geometry(), which refuses anything below 16x16. */ +int xpsb_3d_dest_ok(const struct xpsb_3d_surf *s) +{ + struct xpsb_surface_desc d; + uint32_t fmt; + + if (!s->handle) + return -1; + if (xpsb_3d_format(s->pict_format, &fmt)) + return -1; + if (fmt == XPSB_FMT_YUY2 || fmt == XPSB_FMT_UYVY) + return -1; + /* One pixel, as xpsb_frame_set_geometry() now takes: the render box is + * counted in tiles and a target smaller than one still fills one. + * Refusing below a tile turned every small composite - a glyph, a + * cursor - into a dropped frame. */ + if (s->w < 1 || s->h < 1) + return -1; + + to_desc(&d, s, fmt); + return xpsb_surface_check(&d); +} + +/* xpsb_tex_state() is what xpsb_frame_set_texture() gates on, so asking it + * here is what makes a decision of "3D" survive the binding. */ +int xpsb_3d_tex_ok(const struct xpsb_3d_surf *s, uint32_t fmt) +{ + struct xpsb_surface_desc d; + uint32_t w0, w1; + + if (!s->handle) + return -1; + to_desc(&d, s, fmt); + return xpsb_tex_state(&w0, &w1, &d); +} + +static int comp_tex_ok(const struct xpsb_3d_surf *s) +{ + uint32_t fmt; + + if (xpsb_3d_format(s->pict_format, &fmt)) + return -1; + return xpsb_3d_tex_ok(s, fmt); +} + +/* The four flags xpsb_composite_shader() branches on, derived from the source + * format the same way psb3DPrepareComposite derives them. */ +static int comp_flags(struct xpsb_shader_flags *fl, const struct xpsb_3d_comp *r) +{ + struct xpsb_format f; + uint32_t fmt = 0; + + memset(fl, 0, sizeof *fl); + fl->scalar_src = r->scalar_src ? 1 : 0; + fl->scalar_mask = r->scalar_mask ? 1 : 0; + if (r->scalar_src) + return 0; + + if (xpsb_format_parse(r->src.pict_format, &f) || + xpsb_3d_format(r->src.pict_format, &fmt)) + return -1; + fl->src_no_alpha = !f.has_alpha; + fl->src_is_a8 = fmt == XPSB_FMT_A8; + return 0; +} + +int xpsb_3d_decide_composite(const struct xpsb_3d_comp *r, unsigned caps) +{ + const unsigned need = XPSB_3D_CAP_QUAD | XPSB_3D_CAP_COMPOSITE; + + if ((caps & need) != need) + return 0; + if (r->op < 0 || r->op >= XPSB_NUM_OPS) + return 0; + /* Two constants leave no sampled texture at all, and the relocation + * table has never been generated for zero texture units. */ + if (r->scalar_src && r->scalar_mask) + return 0; + if ((r->scalar_src || r->scalar_mask) && !(caps & XPSB_3D_CAP_SCALAR)) + return 0; + if (xpsb_3d_dest_ok(&r->dst)) + return 0; + + if (!r->scalar_src && (!r->have_src || comp_tex_ok(&r->src))) + return 0; + if (!r->scalar_mask) { + /* A second sampled unit is only needed when the source is a + * texture as well; a scalar source leaves the mask on unit 0. */ + if (!r->scalar_src && !(caps & XPSB_3D_CAP_MASK_TEX)) + return 0; + if (!r->have_mask || comp_tex_ok(&r->mask)) + return 0; + } + return 1; +} + +uint32_t xpsb_3d_plane_format(uint32_t fourcc, unsigned i) +{ + if (i >= (unsigned)xpsb_yuv_planes(fourcc)) + return ~0u; + + switch (fourcc) { + case XPSB_FOURCC_YUY2: return XPSB_FMT_YUY2; + case XPSB_FOURCC_UYVY: return XPSB_FMT_UYVY; + default: return XPSB_FMT_A8; + } +} + +/* psbBlitYUV passes the box, not the surface, so the render target's width + * comes from the stride - the same derivation psbExaSrfInfo makes - and its + * height only has to cover the box, since no row below it is rasterised. */ +int xpsb_3d_video_dest(struct xpsb_3d_surf *out, const struct xpsb_3d_video *r) +{ + struct xpsb_surface_desc d; + uint32_t fmt, bpp; + + if (!r->dst.handle) + return -1; + if (xpsb_3d_format(r->dst.pict_format, &fmt)) + return -1; + if (fmt == XPSB_FMT_YUY2 || fmt == XPSB_FMT_UYVY) + return -1; + + bpp = xpsb_format_bpp(fmt); + if (!bpp || !r->dst.stride || r->dst.stride % bpp) + return -1; + + *out = r->dst; + out->w = r->dst.stride / bpp; + out->h = r->y + r->dst.h; + if (out->w < 16 || out->h < 16) + return -1; + + to_desc(&d, out, fmt); + if (xpsb_surface_check(&d)) + return -1; + return 0; +} + +int xpsb_3d_decide_video(const struct xpsb_3d_video *r, unsigned caps) +{ + const unsigned need = XPSB_3D_CAP_QUAD | XPSB_3D_CAP_VIDEO; + struct xpsb_3d_surf dest; + unsigned i; + + if ((caps & need) != need) + return 0; + if (!xpsb_vidshader(r->fourcc, NULL)) + return 0; + if (!r->nplanes || r->nplanes != (unsigned)xpsb_yuv_planes(r->fourcc)) + return 0; + /* Only the packed FOURCCs are reachable, and this rule is their own + * rather than MASK_TEX's: the planar programs are undecoded FIRH + * instructions on two or three sampled units, reading coefficient + * registers nothing in this tree can name, and their chroma plane has + * a texture format code xpsb_format_bpp() would reject + * (disasm/video-integration.md section 4). */ + if (r->nplanes != 1) + return 0; + /* Without the conversion floats the shader converts with whatever the + * secondary attribute bank held. */ + if (!r->conv) + return 0; + if (!r->dst.w || !r->dst.h) + return 0; + if (xpsb_3d_video_dest(&dest, r)) + return 0; + + for (i = 0; i < r->nplanes; i++) + if (xpsb_3d_tex_ok(&r->plane[i], + xpsb_3d_plane_format(r->fourcc, i))) + return 0; + return 1; +} + +/* ---- the step xpsb_frame.h does not export ---- */ + +/* The packed route only. xpsb_frame_set_video() keeps the plumbing every + * stream this layer has built - both dependency bits, DOUTI 0x0fc0aa00, one + * sampled unit, the eleven-float vertex record - and moves the shader's three + * texel reads from pa0 to pa1 instead, which is where the sample lands when a + * colour is iterated. Nothing that has already submitted changes. + * + * Two words in the result are not captured in this configuration and are the + * first to suspect if a frame comes out wrong: the temporary-register count + * (heap 0xd1, transcribed from psbBlitYUV but never seen alongside iterated + * varyings) and the three moved shader words. Both are wrong-colour risks + * rather than submission risks - see disasm/video-integration.md sections 1 + * and 3. If the image turns out to be the converted quad colour rather than + * converted video, that document's Route A is the fallback. + * + * The planar formats stay on the CPU; xpsb_3d_decide_video() has already + * refused them, and this refuses them again rather than rely on that. + * + * Reproduced deliberately: the program's three packs write bytes 2, 1 and 0 of + * o0 and leave the alpha byte at whatever the previous task left, exactly as + * the closed module does. Invisible for an x8r8g8b8 destination, undefined for + * an a8r8g8b8 one. */ +static int frame_set_vidshader(struct xpsb_frame *f, + const struct xpsb_3d_video *r) +{ + if (r->nplanes != 1 || !r->conv) { + errno = ENOSYS; + return -1; + } + return xpsb_frame_set_video(f, r->fourcc, r->conv); +} + +/* ---- execution ---- */ + +int xpsb_3d_init(struct xpsb_3d *g, int drmfd, int w, int h) +{ + memset(g, 0, sizeof *g); + g->caps = xpsb_3d_caps(); + if (!g->caps) + return -1; + + g->frame = calloc(1, sizeof *g->frame); + if (!g->frame) + return -1; + if (xpsb_frame_init(g->frame, drmfd, w, h)) { + int e = errno; + + free(g->frame); + g->frame = NULL; + errno = e; + return -1; + } + g->ready = 1; + return 0; +} + +void xpsb_3d_fini(struct xpsb_3d *g) +{ + if (g->ready) + xpsb_frame_fini(g->frame); + free(g->frame); + g->frame = NULL; + g->ready = g->active = g->staged = 0; +} + +/* Geometry first: xpsb_frame_set_geometry() rebuilds the streams that depend + * on the render size and resets the destination to the frame's own buffer, so + * everything else has to follow it. Depth testing goes off - the depth buffer + * is only ever initialised by the clearing draws this path drops. */ +static int frame_begin_dest(struct xpsb_3d *g, const struct xpsb_3d_surf *s) +{ + struct xpsb_surface_desc d; + uint32_t fmt; + + if (xpsb_3d_format(s->pict_format, &fmt)) + return -1; + to_desc(&d, s, fmt); + + if (xpsb_frame_set_geometry(g->frame, (int)d.w, (int)d.h, + d.stride / xpsb_format_bpp(fmt), 0)) + return -1; + /* Deliberate: the destination has to survive outside the quad, so the + * two clearing draws go and with them ten relocations. No submission + * has ever carried the stream this leaves - see xpsb_frame.c. */ + if (xpsb_frame_set_clear(g->frame, 0)) + return -1; + return xpsb_frame_set_dest(g->frame, &d); +} + +static int bind_tex(struct xpsb_3d *g, int unit, const struct xpsb_3d_surf *s) +{ + struct xpsb_surface_desc t; + uint32_t fmt; + + if (xpsb_3d_format(s->pict_format, &fmt)) + return -1; + to_desc(&t, s, fmt); + return xpsb_frame_set_texture(g->frame, unit, &t); +} + +int xpsb_3d_composite_begin(struct xpsb_3d *g, const struct xpsb_3d_comp *r) +{ + struct xpsb_shader_flags fl; + int scalar = r->scalar_src || r->scalar_mask; + + g->active = g->staged = 0; + if (!g->ready || !xpsb_3d_decide_composite(r, g->caps)) + return -1; + if (comp_flags(&fl, r)) + return -1; + /* A preceding video blit left the USE task on the YUV program and the + * whole secondary attribute bank on its conversion floats. */ + if (xpsb_frame_set_video(g->frame, 0, NULL)) + return -1; + + if (xpsb_frame_set_ntex(g->frame, scalar ? 1 : 2)) + return -1; + if (frame_begin_dest(g, &r->dst)) + return -1; + + /* The modulate's texture operand is pa0 whichever of the two it is - + * sop2 i0, pa0, sa0 for a scalar mask and sop2 i0, sa0, pa0 for a + * scalar source - so the one sampled unit of a scalar composite is + * unit 0, and psb3DCompositeQuad's coordinates already follow it. */ + if (bind_tex(g, 0, r->scalar_src ? &r->mask : &r->src)) + return -1; + if (!scalar && bind_tex(g, 1, &r->mask)) + return -1; + if (xpsb_frame_set_scalar(g->frame, scalar, r->scalar)) + return -1; + if (xpsb_frame_set_composite(g->frame, r->op, &fl)) + return -1; + + g->active = 1; + return 0; +} + +static int stage_quad(struct xpsb_3d *g, const struct xpsb_3d_quad *q) +{ + struct xpsb_quad fq; + + memset(&fq, 0, sizeof fq); + fq.x0 = q->x0; fq.y0 = q->y0; fq.x1 = q->x1; fq.y1 = q->y1; + fq.u0 = q->u0; fq.v0 = q->v0; fq.u1 = q->u1; fq.v1 = q->v1; + fq.r = fq.g = fq.b = fq.a = 1.0f; + + if (xpsb_frame_set_quad(g->frame, &fq)) + return -1; + g->staged = 1; + return 0; +} + +static int flush(struct xpsb_3d *g) +{ + if (!g->staged) + return 0; + g->staged = 0; + return xpsb_frame_submit(g->frame); +} + +/* A frame holds one quad, so a second one submits the first rather than + * overwriting it. EXA calls this repeatedly between prepare and finish. */ +int xpsb_3d_composite_quad(struct xpsb_3d *g, const struct xpsb_3d_quad *q) +{ + if (!g->active) + return -1; + if (flush(g)) + return -1; + return stage_quad(g, q); +} + +int xpsb_3d_composite_finish(struct xpsb_3d *g) +{ + int ret = g->active ? flush(g) : -1; + + g->active = 0; + return ret; +} + +int xpsb_3d_video_blit(struct xpsb_3d *g, const struct xpsb_3d_video *r) +{ + struct xpsb_3d_surf dest; + struct xpsb_surface_desc t; + struct xpsb_3d_quad q; + unsigned i; + + if (!g->ready || !xpsb_3d_decide_video(r, g->caps)) + return -1; + if (xpsb_3d_video_dest(&dest, r)) + return -1; + + /* A preceding composite may have left a second sampled unit and the + * blend enable behind; the video quad replaces the destination and its + * shader computes no blend factors. */ + if (xpsb_frame_set_ntex(g->frame, r->nplanes)) + return -1; + if (xpsb_frame_set_blend(g->frame, 0)) + return -1; + if (frame_begin_dest(g, &dest)) + return -1; + + for (i = 0; i < r->nplanes; i++) { + to_desc(&t, &r->plane[i], + xpsb_3d_plane_format(r->fourcc, i)); + if (xpsb_frame_set_texture(g->frame, (int)i, &t)) + return -1; + } + if (frame_set_vidshader(g->frame, r)) + return -1; + + q.x0 = (float)r->x; + q.y0 = (float)r->y; + q.x1 = (float)(r->x + r->dst.w); + q.y1 = (float)(r->y + r->dst.h); + q.u0 = r->u0; q.v0 = r->v0; q.u1 = r->u1; q.v1 = r->v1; + + if (stage_quad(g, &q)) + return -1; + return flush(g); +} diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/xpsb_3d.h mesa-26.2.2/src/gallium/drivers/sgx/xpsb_3d.h --- mesa-26.2.2.orig/src/gallium/drivers/sgx/xpsb_3d.h 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/xpsb_3d.h 2026-09-08 10:57:36.684620133 +0200 @@ -0,0 +1,141 @@ +/* Open replacement for Xpsb.so - choosing between the SGX and the CPU path. + * + * The entry points in xpsb_module.c can carry an operation out in two ways: + * by building a frame for the SGX (xpsb_frame.c) or by mapping the buffer + * objects and evaluating it (xpsb_comp.c, xpsb_yuv.c). This file owns the + * choice and the frame-side execution; the CPU path stays reachable for + * everything the frame layer declines, and both must produce the same result. + * + * The choice is a pure function of the operation - xpsb_3d_decide_composite() + * and xpsb_3d_decide_video() touch no device, so test_3d.c can exercise every + * rule on the host. They are built from the same predicates the frame's own + * setters apply, so a decision of "3D" cannot be contradicted afterwards by + * anything except an ioctl failure. + * + * Neither xpsb_frame.h nor Xpsb.h is included here: xpsb_module.c includes + * xpsb_comp.h, which declares a different struct xpsb_quad, and the two + * headers cannot meet in one translation unit. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#ifndef _XPSB_3D_H_ +#define _XPSB_3D_H_ + +#include + +struct xpsb_frame; + +/* What the frame layer can be asked for. A bit that is clear sends every + * operation needing it to the CPU path; see xpsb_3d_caps() for which of them + * xpsb_frame.h actually exports today. */ +#define XPSB_3D_CAP_QUAD 0x1u /* destination, one texture, one quad */ +#define XPSB_3D_CAP_COMPOSITE 0x2u /* operator, blending, composite shader */ +#define XPSB_3D_CAP_MASK_TEX 0x4u /* a second sampled texture unit */ +#define XPSB_3D_CAP_SCALAR 0x8u /* a constant operand in a secondary + attribute, for a scalar source or + a scalar mask */ +#define XPSB_3D_CAP_VIDEO 0x10u /* the YUV programs and their coefficients */ + +#define XPSB_3D_MAX_PLANES 3 + +/* psbSetupConversionData's array: a row-major 3x3 matrix, the luma bias, and + * videoGamma. The packed shader reads the first ten as sa0..sa9. */ +#define XPSB_3D_CONV_N 11 + +/* An XpsbSurface reduced to what the decision and the frame need. Filters and + * addressing modes use the XpsbFilterFormats / XpsbAddrModes numbering, which + * xpsb_frame.h shares. handle is the DRM buffer object; zero is no buffer. */ +struct xpsb_3d_surf { + uint32_t handle, offset, pict_format; + uint32_t w, h, stride; + uint32_t umode, vmode, minfilter, magfilter; +}; + +/* One prepared composite. dst carries the whole destination surface, as + * psb3DPrepareComposite passes it; quad coordinates are absolute in it. */ +struct xpsb_3d_comp { + int op; + int scalar_src, scalar_mask; + int have_src, have_mask; + uint32_t scalar; /* psbPixelARGB8888's word, in sa0 */ + struct xpsb_3d_surf dst, src, mask; +}; + +/* One video blit. Unlike the composite destination, dst here is the *box* + * psbBlitYUV writes - x, y, w, h - so the render target's own width and + * height have to be inferred; see xpsb_3d_video_dest(). The plane surfaces' + * pict_format is ignored: the hardware format follows from the FOURCC. */ +struct xpsb_3d_video { + uint32_t fourcc; + unsigned nplanes; + uint32_t x, y; + struct xpsb_3d_surf dst, plane[XPSB_3D_MAX_PLANES]; + float u0, v0, u1, v1; + /* The XPSB_3D_CONV_N conversion floats, or NULL. NULL declines: the + * shader would convert with whatever the attribute bank held, and a + * wrong render is worse than the CPU path. Only meaningful for a packed + * FOURCC - psbBlitYUV overloads the same argument for the planar ones, + * where it is nine fixed-point words rather than floats. */ + const float *conv; +}; + +/* Destination pixel coordinates, normalised texture coordinates. */ +struct xpsb_3d_quad { + float x0, y0, x1, y1; + float u0, v0, u1, v1; +}; + +struct xpsb_3d { + struct xpsb_frame *frame; + unsigned caps; + int ready; /* the frame's buffer objects exist */ + int active; /* the running composite is on the 3D path */ + int staged; /* a quad is in the frame, not submitted */ +}; + +/* The nine buffer objects a frame owns are allocated here and nowhere else, + * so this belongs in XpsbInit and its counterpart in XpsbTakeDown. A failure + * is not fatal: it leaves ready clear and every operation on the CPU path. */ +int xpsb_3d_init(struct xpsb_3d *g, int drmfd, int w, int h); +void xpsb_3d_fini(struct xpsb_3d *g); + +/* The capability set, XPSB_NO_3D honoured: any value other than "0" in the + * environment returns zero and forces the CPU path everywhere. */ +unsigned xpsb_3d_caps(void); + +/* Pure predicates, all device-free. + * + * xpsb_3d_format maps a PICT format onto the hardware's five-bit surface code + * and returns -1 when there is none. The two _ok predicates return 0 when the + * surface can be used and -1 otherwise, and the two decide functions return 1 + * for the 3D path and 0 for the CPU path. */ +int xpsb_3d_format(uint32_t pict_format, uint32_t *fmt); +int xpsb_3d_dest_ok(const struct xpsb_3d_surf *s); +int xpsb_3d_tex_ok(const struct xpsb_3d_surf *s, uint32_t fmt); +int xpsb_3d_decide_composite(const struct xpsb_3d_comp *r, unsigned caps); +int xpsb_3d_decide_video(const struct xpsb_3d_video *r, unsigned caps); + +/* Hardware format of video plane i, or ~0u for a FOURCC without one. */ +uint32_t xpsb_3d_plane_format(uint32_t fourcc, unsigned i); + +/* The render target a video blit writes into, derived from the box: the width + * from the stride, the height from y + h. Returns -1 if that does not yield a + * surface the hardware can address. */ +int xpsb_3d_video_dest(struct xpsb_3d_surf *out, const struct xpsb_3d_video *r); + +/* Execution. Every one of these returns 0 on success and -1 when the caller + * has to use the CPU path instead - the frame layer's convention, not + * psb3DPrepareComposite's. + * + * xpsb_3d_composite_begin() makes the decision final: after it returns 0 the + * operation is on the 3D path until xpsb_3d_composite_finish(), which is what + * lets psb3DCompositeQuad return void. */ +int xpsb_3d_composite_begin(struct xpsb_3d *g, const struct xpsb_3d_comp *r); +int xpsb_3d_composite_quad(struct xpsb_3d *g, const struct xpsb_3d_quad *q); +int xpsb_3d_composite_finish(struct xpsb_3d *g); + +int xpsb_3d_video_blit(struct xpsb_3d *g, const struct xpsb_3d_video *r); + +#endif /* _XPSB_3D_H_ */ diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/xpsb_comp.c mesa-26.2.2/src/gallium/drivers/sgx/xpsb_comp.c --- mesa-26.2.2.orig/src/gallium/drivers/sgx/xpsb_comp.c 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/xpsb_comp.c 2026-09-08 10:57:36.684627964 +0200 @@ -0,0 +1,154 @@ +/* Open replacement for Xpsb.so - Render composite execution. + * + * The factor pairs below are the same ones the generated pixel shader + * applies, see xpsb_shader.c; this file evaluates them on the CPU so that the + * operation completes without a 3D submission. Colours are premultiplied + * throughout, as the Render protocol requires. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#include + +#include "xpsb_comp.h" +#include "xpsb_shader.h" + +int xpsb_op_is_source_only(int op) +{ + return op >= 0 && op < XPSB_NUM_OPS && xpsb_opaque_ops[op]; +} + +static void blend_factors(int op, int sa, int da, int *fs, int *fd) +{ + switch (op) { + case 0: *fs = 0; *fd = 0; break; /* Clear */ + case 1: *fs = 255; *fd = 0; break; /* Src */ + case 2: *fs = 0; *fd = 255; break; /* Dst */ + case 3: *fs = 255; *fd = 255 - sa; break; /* Over */ + case 4: *fs = 255 - da; *fd = 255; break; /* OverReverse */ + case 5: *fs = da; *fd = 0; break; /* In */ + case 6: *fs = 0; *fd = sa; break; /* InReverse */ + case 7: *fs = 255 - da; *fd = 0; break; /* Out */ + case 8: *fs = 0; *fd = 255 - sa; break; /* OutReverse */ + case 9: *fs = da; *fd = 255 - sa; break; /* Atop */ + case 10: *fs = 255 - da; *fd = sa; break; /* AtopReverse */ + case 11: *fs = 255 - da; *fd = 255 - sa; break; /* Xor */ + case 12: *fs = 255; *fd = 255; break; /* Add */ + default: /* Saturate */ + *fs = sa ? ((255 - da) * 255) / sa : 255; + if (*fs > 255) + *fs = 255; + *fd = 255; + break; + } +} + +static unsigned int mul255(unsigned int a, unsigned int b) +{ + unsigned int t = a * b + 128; + + return (t + (t >> 8)) >> 8; +} + +static uint32_t blend_pixel(int op, uint32_t src, uint32_t dst) +{ + int fs, fd, i; + uint32_t out = 0; + + blend_factors(op, (int)(src >> 24), (int)(dst >> 24), &fs, &fd); + + for (i = 0; i < 4; i++) { + int shift = 8 * i; + unsigned int cs = (src >> shift) & 0xff; + unsigned int cd = (dst >> shift) & 0xff; + unsigned int v = mul255(cs, (unsigned int)fs) + + mul255(cd, (unsigned int)fd); + + if (v > 255) + v = 255; + out |= v << shift; + } + return out; +} + +/* Modulate the source by the mask alpha; the shader has no component-alpha + * path, so only the mask's alpha takes part. */ +static uint32_t modulate(uint32_t src, unsigned int ma) +{ + uint32_t out = 0; + int i; + + if (ma == 0xff) + return src; + for (i = 0; i < 4; i++) + out |= mul255((src >> (8 * i)) & 0xff, ma) << (8 * i); + return out; +} + +void xpsb_comp_quad(const struct xpsb_comp_state *st, + const struct xpsb_quad *q) +{ + int x0 = q->x0, y0 = q->y0, x1 = q->x1, y1 = q->y1; + int rw = q->x1 - q->x0, rh = q->y1 - q->y0; + int x, y; + unsigned int dbytes = st->dst.fmt.bpp >> 3; + int source_only = xpsb_op_is_source_only(st->op); + float dsu, dsv, dmu, dmv; + + if (rw <= 0 || rh <= 0) + return; + if (x0 < 0) + x0 = 0; + if (y0 < 0) + y0 = 0; + if (x1 > (int)st->dst.w) + x1 = (int)st->dst.w; + if (y1 > (int)st->dst.h) + y1 = (int)st->dst.h; + + /* Step the texture coordinates rather than reinterpolating, and take + * them at pixel centres so a 1:1 blit lands exactly on texel i. */ + dsu = (q->su1 - q->su0) / (float)rw; + dsv = (q->sv1 - q->sv0) / (float)rh; + dmu = (q->mu1 - q->mu0) / (float)rw; + dmv = (q->mv1 - q->mv0) / (float)rh; + + for (y = y0; y < y1; y++) { + float cy = (float)(y - q->y0) + 0.5f; + float sv = q->sv0 + cy * dsv; + float mv = q->mv0 + cy * dmv; + uint8_t *row = st->dst.base + (ptrdiff_t)y * st->dst.stride; + + for (x = x0; x < x1; x++) { + float cx = (float)(x - q->x0) + 0.5f; + uint8_t *p = row + (ptrdiff_t)x * dbytes; + uint32_t src, dst; + unsigned int ma; + + if (st->scalar_src) + src = st->scalar; + else + src = xpsb_image_sample(&st->src, + q->su0 + cx * dsu, sv); + + /* One scalar field serves both, so a caller that sets + * both flags gets an opaque mask rather than the + * source's own alpha applied twice. */ + if (st->scalar_mask) + ma = st->scalar_src ? 0xff : (st->scalar >> 24); + else + ma = xpsb_image_sample(&st->mask, + q->mu0 + cx * dmu, mv) >> 24; + + src = modulate(src, ma); + + /* Clear and Src do not read the destination, which + * saves a read-back on the uncached mapping. */ + dst = source_only ? 0 : xpsb_pix_load(&st->dst.fmt, p); + + xpsb_pix_store(&st->dst.fmt, p, + blend_pixel(st->op, src, dst)); + } + } +} diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/xpsb_comp.h mesa-26.2.2/src/gallium/drivers/sgx/xpsb_comp.h --- mesa-26.2.2.orig/src/gallium/drivers/sgx/xpsb_comp.h 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/xpsb_comp.h 2026-09-08 10:57:36.684636034 +0200 @@ -0,0 +1,42 @@ +/* Open replacement for Xpsb.so - Render composite execution. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#ifndef _XPSB_COMP_H_ +#define _XPSB_COMP_H_ + +#include "xpsb_pixel.h" + +/* One prepared composite operation. Exactly one of src/scalar_src and one of + * mask/scalar_mask is in use, which is the invariant psbExaPrepareComposite3D + * establishes before it calls us. */ +struct xpsb_comp_state { + int op; + struct xpsb_image dst; + struct xpsb_image src; + struct xpsb_image mask; + int scalar_src; + int scalar_mask; + /* The source colour when scalar_src is set, otherwise the mask pixel + * whose alpha modulates the source - psbExaPrepareComposite3D never + * sets both, and clears scalar_src if it would have. */ + uint32_t scalar; +}; + +/* One axis-aligned destination rectangle with its texture coordinates at the + * top left and bottom right corners. */ +struct xpsb_quad { + int x0, y0, x1, y1; + float su0, sv0, su1, sv1; + float mu0, mv0, mu1, mv1; +}; + +/* Non-zero for operators that never read the destination. */ +int xpsb_op_is_source_only(int op); + +void xpsb_comp_quad(const struct xpsb_comp_state *st, + const struct xpsb_quad *q); + +#endif /* _XPSB_COMP_H_ */ diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/xpsb_frame.c mesa-26.2.2/src/gallium/drivers/sgx/xpsb_frame.c --- mesa-26.2.2.orig/src/gallium/drivers/sgx/xpsb_frame.c 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/xpsb_frame.c 2026-09-08 10:57:36.684677726 +0200 @@ -0,0 +1,4345 @@ +/* Open replacement for Xpsb.so - the SGX frame renderer. + * + * Moved out of test/drmcube.c with the words it emits unchanged. The heap, + * the rastgeom records, the draw records, the four register/value streams and + * the two relocation lists are generated, not replayed; the buffer table is + * carried across as the literal it is there. Constants whose bit-level + * meaning is not decoded are carried verbatim and named by region. + * + * Beyond drmcube: the destination and the sampled texture can be replaced by + * caller-supplied surfaces, the cube mesh by a single quad, and the two + * clearing draws can be dropped. The clear is on by default because that path + * is the one that renders on this silicon today; nothing here has run without + * it. test_frame.c holds every buffer to drmcube's output dword for dword + * wherever behaviour should not have changed. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#ifndef _GNU_SOURCE +#define _GNU_SOURCE +#endif +#define _LARGEFILE64_SOURCE +#include +#include +#include +#include +#include +#include +#include + +#include "xpsb_frame.h" +#include "xpsb_heap.h" +#include "xpsb_pds.h" + +/* getenv() is a linear scan of the environment and glibc's does a strncmp per + * entry. These are asked per issue list and per record, which measured at a + * large part of a frame's CPU; the answer cannot change during a run. */ +#define XPSB_ENVS(name) __extension__({ \ + static const char *xpsb_envs_cached_; \ + static int xpsb_envs_got_; \ + if (!xpsb_envs_got_) { \ + xpsb_envs_cached_ = getenv(name); \ + xpsb_envs_got_ = 1; \ + } \ + xpsb_envs_cached_; \ +}) + + +/* ---- psb / DRM ABI ---- + * + * Mirrored from the GPL kernel module's drm.h and psb_drm.h. The command + * argument is the form without PSB_DETEAR, which is what drmcube submits and + * what the running kernel accepts; DRM dispatches on the ioctl number alone, + * so the size encoded here only has to cover the fields the driver reads. */ +#define DRM_IOCTL_BASE 'd' +#define DRM_COMMAND_BASE 0x40 +#define DRM_IOWR(nr, sz) ((3u << 30) | (DRM_IOCTL_BASE << 8) | (nr) | ((sz) << 16)) + +#define DRM_BO_FLAG_READ (1ULL << 0) +#define DRM_BO_FLAG_WRITE (1ULL << 1) +#define DRM_BO_FLAG_MAPPABLE (1ULL << 5) +#define DRM_BO_FLAG_MEM_LOCAL (1ULL << 24) +#define DRM_BO_FLAG_MEM_TT (1ULL << 25) + +struct bo_create_req { uint64_t mask, size, buffer_start; unsigned hint, page_alignment; }; +struct bo_info_rep { + uint64_t flags, mask, size, offset, arg_handle, buffer_start; + unsigned handle, fence_flags, rep_flags, page_alignment; + unsigned desired_tile_stride, hw_tile_stride, tile_info, pad64; + uint64_t expand_pad[4]; +}; +struct bo_info_req { + uint64_t mask, flags; + unsigned handle, hint, fence_class, desired_tile_stride, tile_info, pad64; + uint64_t presumed_offset; +}; +union bo_create_arg { struct bo_create_req req; struct bo_info_rep rep; }; +union bo_map_arg { struct bo_info_req req; struct bo_info_rep rep; }; +struct bo_handle_arg { unsigned handle; }; +struct bo_arg_rep { struct bo_info_rep bo_info; int ret; unsigned pad; }; +struct bo_op_req { int op; unsigned arg_handle; struct bo_info_req bo_req; }; +struct bo_op_arg { + uint64_t next; + union { struct bo_op_req req; struct bo_arg_rep rep; } d; + int handled; unsigned pad64; +}; +struct fence_arg { + unsigned handle, fence_class, type, flags, signaled, error, sequence, pad64; + uint64_t expand_pad[2]; +}; +/* The trailing sVideoInfo is the PSB_DETEAR form, 144 bytes rather than 116. + * drm_unlocked_ioctl copies _IOC_SIZE(cmd) bytes into a fixed 512-byte stack + * buffer and its size check is #if 0'd out, so the short form is accepted - + * but it leaves sVideoInfo as uninitialised kernel stack. psb_cmdbuf_2d reads + * that flag, and a value of PSB_DELAYED_2D_BLIT makes the kernel spin waiting + * for a display interrupt that will never come. Engines 2 and 3 never reach + * that read, which is why the short form worked; sending the full form zeroed + * is safe against either kernel build and is required once anything here + * submits on engine 0. */ +struct psb_video_info { + uint32_t flag; + uint32_t x, y, w, h; + uint32_t pFBBOHandle; + void *pFBVirtAddr; +}; +struct psb_cmdbuf_arg { + uint64_t buffer_list, clip_rects, scene_arg, fence_arg; + uint32_t ta_flags, ta_handle, ta_offset, ta_size; + uint32_t oom_handle, oom_offset, oom_size; + uint32_t cmdbuf_handle, cmdbuf_offset, cmdbuf_size; + uint32_t reloc_handle, reloc_offset, num_relocs; + int32_t damage; uint32_t fence_flags, engine; + uint32_t feedback_ops, feedback_handle, feedback_offset; + uint32_t feedback_breakpoints, feedback_size; + struct psb_video_info sVideoInfo; +}; + +#define IOC_BO_CREATE DRM_IOWR(0xcd, sizeof(union bo_create_arg)) +#define IOC_BO_MAP DRM_IOWR(0xcf, sizeof(union bo_map_arg)) +#define IOC_BO_UNMAP DRM_IOWR(0xd0, sizeof(struct bo_handle_arg)) +#define IOC_BO_UNREFERENCE DRM_IOWR(0xd2, sizeof(union bo_map_arg)) +#define IOC_FENCE_WAIT DRM_IOWR(0xca, sizeof(struct fence_arg)) +#define IOC_FENCE_UNREF DRM_IOWR(0xc7, sizeof(struct fence_arg)) +#define IOC_PSB_CMDBUF DRM_IOWR(DRM_COMMAND_BASE + 0x00, sizeof(struct psb_cmdbuf_arg)) + +/* ---- the captured buffer set ---- */ +const struct xpsb_fb_desc xpsb_frame_bufs[XPSB_FRAME_NBUF] = { + { 7864320ull, 0x0000000020000021ull, 0x0000000000ull, 0x0000000000ull }, /* 0 */ + { 65536ull, 0x0000000004000023ull, 0x0016000000ull, 0x00ff000000ull }, /* 1 */ + { 131072ull, 0x0008000020000021ull, 0x0000000000ull, 0x0000000000ull }, /* 2 */ + { 98304ull, 0x0000000080000021ull, 0x0000000000ull, 0x0000000000ull }, /* 3 */ + { 2097152ull, 0x00000000010000a4ull, 0x0000000000ull, 0x0000000000ull }, /* 4 */ + { 4554752ull, 0x0001000010000021ull, 0x0000000000ull, 0x0000000000ull }, /* 5 */ + { 65536ull, 0x0000000004000023ull, 0x0000000000ull, 0x0000000000ull }, /* 6 */ + { 16384ull, 0x0000000004000023ull, 0x0000000000ull, 0x0000000000ull }, /* 7 */ +}; +const int xpsb_ta_list[XPSB_FRAME_NBUF] = { 0, 1, 2, 3, 4, 5, 6, 7 }; +const int xpsb_raster_list[XPSB_RAS_LIST_LEN] = { 0, -1, 2, 1, 3, 4 }; +/* per-pass validate flags for the raster list (may differ from TA) */ +const uint64_t xpsb_raster_vflags[XPSB_RAS_LIST_LEN] = { + 0x0ull, 0x16000000ull, 0x0ull, 0x16000000ull, 0x0ull, 0x0ull +}; +const uint64_t xpsb_raster_vmask[XPSB_RAS_LIST_LEN] = { + 0x0ull, 0xff000000ull, 0x0ull, 0xff000000ull, 0x0ull, 0x0ull +}; + +/* ---- relocation descriptors, carried from test/reloc_tables.h ---- */ +const struct xpsb_reloc xpsb_ta_relocs[XPSB_NUM_TA_RELOCS] = { + /* op, where, buffer, mask, shift, pre_add, background, dst_buffer, arg0, arg1 */ + { XPSB_RELOC_OP_OFFSET , 0x000001, 1, 0xfffffffc, 0x20002, 0x0, 0x0, 0, 0x0, 0 }, + { XPSB_RELOC_OP_OFFSET , 0x00000a, 0, 0xffffffff, 0x0, 0x0, 0x0, 0, 0x0, 0 }, + { XPSB_RELOC_OP_USE_REG , 0x00000c, 2, 0x0000000f, 0x0, 0x0, 0x200000, 0, 0x10, 1 }, + { XPSB_RELOC_OP_USE_OFFSET , 0x00000c, 2, 0x000000f0, 0xf0004, 0x0, 0x0, 0, 0x10, 1 }, + { XPSB_RELOC_OP_USE_OFFSET , 0x00000c, 2, 0x0007ff00, 0x40008, 0x0, 0x0, 0, 0x10, 1 }, + { XPSB_RELOC_OP_USE_REG , 0x000008, 2, 0x0000000f, 0x0, 0x20, 0x0, 0, 0x8, 1 }, + { XPSB_RELOC_OP_USE_OFFSET , 0x000008, 2, 0x000000f0, 0xf0004, 0x20, 0x0, 0, 0x8, 1 }, + { XPSB_RELOC_OP_USE_OFFSET , 0x000008, 2, 0x0007ff00, 0x40008, 0x20, 0x0, 0, 0x8, 1 }, + { XPSB_RELOC_OP_USE_REG , 0x000030, 2, 0x0000000f, 0x0, 0x40, 0x0, 0, 0x8, 1 }, + { XPSB_RELOC_OP_USE_OFFSET , 0x000030, 2, 0x000000f0, 0xf0004, 0x40, 0x0, 0, 0x8, 1 }, + { XPSB_RELOC_OP_USE_OFFSET , 0x000030, 2, 0x0007ff00, 0x40008, 0x40, 0x0, 0, 0x8, 1 }, + { XPSB_RELOC_OP_OFFSET , 0x000000, 0, 0x00ffffff, 0x40000, 0x100, 0x0, 3, 0x0, 0 }, + { XPSB_RELOC_OP_OFFSET , 0x000002, 0, 0x00ffffff, 0x40000, 0xc0, 0xc000000, 3, 0x0, 0 }, + { XPSB_RELOC_OP_USE_REG , 0x000048, 2, 0x0000000f, 0x0, 0x60, 0x100000, 0, 0x8, 1 }, + { XPSB_RELOC_OP_USE_OFFSET , 0x000048, 2, 0x000000f0, 0xf0004, 0x60, 0x0, 0, 0x8, 1 }, + { XPSB_RELOC_OP_USE_OFFSET , 0x000048, 2, 0x0007ff00, 0x40008, 0x60, 0x0, 0, 0x8, 1 }, + { XPSB_RELOC_OP_OFFSET , 0x000052, 1, 0xffffffff, 0x0, 0x0, 0x0, 0, 0x0, 0 }, + { XPSB_RELOC_OP_OFFSET , 0x000014, 0, 0x00ffffff, 0x40000, 0x160, 0x0, 3, 0x0, 0 }, + { XPSB_RELOC_OP_OFFSET , 0x000016, 0, 0x00ffffff, 0x40000, 0x120, 0xc000000, 3, 0x0, 0 }, + { XPSB_RELOC_OP_OFFSET , 0x000060, 0, 0xffffffff, 0x0, 0x164, 0x0, 0, 0x0, 0 }, + { XPSB_RELOC_OP_USE_REG , 0x000062, 2, 0x0000000f, 0x0, 0x80, 0x200000, 0, 0x10, 0 }, + { XPSB_RELOC_OP_USE_OFFSET , 0x000062, 2, 0x000000f0, 0xf0004, 0x80, 0x0, 0, 0x10, 0 }, + { XPSB_RELOC_OP_USE_OFFSET , 0x000062, 2, 0x0007ff00, 0x40008, 0x80, 0x0, 0, 0x10, 0 }, + { XPSB_RELOC_OP_OFFSET , 0x00000d, 5, 0xffffffff, 0x0, 0x0, 0x0, 4, 0x0, 0 }, + { XPSB_RELOC_OP_OFFSET , 0x000025, 6, 0xf0000000, 0x0, 0x0, 0x0, 4, 0x0, 0 }, + { XPSB_RELOC_OP_OFFSET , 0x000027, 6, 0x0ffffff0, 0x0, 0x0, 0x0, 4, 0x0, 0 }, + { XPSB_RELOC_OP_OFFSET , 0x000029, 6, 0x0ffffff0, 0x0, 0x0, 0x0, 4, 0x0, 0 }, + { XPSB_RELOC_OP_OFFSET , 0x00002b, 6, 0x0ffffff0, 0x0, 0x0, 0x0, 4, 0x0, 0 }, + { XPSB_RELOC_OP_OFFSET , 0x00002d, 6, 0x0ffffff0, 0x0, 0x0, 0x0, 4, 0x0, 0 }, + { XPSB_RELOC_OP_OFFSET , 0x000031, 3, 0x00ffffff, 0x40000, 0x0, 0x2000000, 4, 0x0, 0 }, + { XPSB_RELOC_OP_OFFSET , 0x00003d, 0, 0xffffffff, 0x40004, 0x20, 0x0, 4, 0x0, 0 }, + { XPSB_RELOC_OP_OFFSET , 0x000043, 3, 0x00ffffff, 0x40000, 0x50, 0x2000000, 4, 0x0, 0 }, + { XPSB_RELOC_OP_OFFSET , 0x000088, 0, 0xffffffff, 0x0, 0x1bc, 0x0, 0, 0x0, 0 }, + { XPSB_RELOC_OP_USE_REG , 0x00008c, 2, 0x0000000f, 0x0, 0xa0, 0x200000, 0, 0x10, 0 }, + { XPSB_RELOC_OP_USE_OFFSET , 0x00008c, 2, 0x000000f0, 0xf0004, 0xa0, 0x0, 0, 0x10, 0 }, + { XPSB_RELOC_OP_USE_OFFSET , 0x00008c, 2, 0x0007ff00, 0x40008, 0xa0, 0x0, 0, 0x10, 0 }, + { XPSB_RELOC_OP_USE_REG , 0x0000a0, 2, 0x0000000f, 0x0, 0xc0, 0x0, 0, 0x8, 1 }, + { XPSB_RELOC_OP_USE_OFFSET , 0x0000a0, 2, 0x000000f0, 0xf0004, 0xc0, 0x0, 0, 0x8, 1 }, + { XPSB_RELOC_OP_USE_OFFSET , 0x0000a0, 2, 0x0007ff00, 0x40008, 0xc0, 0x0, 0, 0x8, 1 }, + { XPSB_RELOC_OP_OFFSET , 0x0000b4, 0, 0x00ffffff, 0x40000, 0x2c0, 0x0, 0, 0x0, 0 }, + { XPSB_RELOC_OP_OFFSET , 0x0000b6, 0, 0x00ffffff, 0x40000, 0x280, 0xc000000, 0, 0x0, 0 }, + { XPSB_RELOC_OP_OFFSET , 0x0000c0, 0, 0xffffffff, 0x0, 0x2c4, 0x0, 0, 0x0, 0 }, + { XPSB_RELOC_OP_USE_REG , 0x0000c2, 2, 0x0000000f, 0x0, 0xe0, 0x200000, 0, 0x10, 0 }, + { XPSB_RELOC_OP_USE_OFFSET , 0x0000c2, 2, 0x000000f0, 0xf0004, 0xe0, 0x0, 0, 0x10, 0 }, + { XPSB_RELOC_OP_USE_OFFSET , 0x0000c2, 2, 0x0007ff00, 0x40008, 0xe0, 0x0, 0, 0x10, 0 }, + { XPSB_RELOC_OP_OFFSET , 0x000000, 0, 0x0fffffff, 0x40000, 0x300, 0x40000000, 5, 0x0, 0 }, + { XPSB_RELOC_OP_OFFSET , 0x000003, 0, 0xffffffff, 0x0, 0x1fc, 0x0, 5, 0x0, 0 }, + { XPSB_RELOC_OP_OFFSET , 0x000005, 0, 0x0fffffff, 0x40000, 0x220, 0x0, 5, 0x0, 0 }, + { XPSB_RELOC_OP_OFFSET , 0x0000da, 7, 0xffffffff, 0x0, 0x0, 0x0, 0, 0x0, 0 }, + { XPSB_RELOC_OP_USE_REG , 0x0000d0, 2, 0x0000000f, 0x0, 0x100, 0x180000, 0, 0x8, 1 }, + { XPSB_RELOC_OP_USE_OFFSET , 0x0000d0, 2, 0x000000f0, 0xf0004, 0x100, 0x0, 0, 0x8, 1 }, + { XPSB_RELOC_OP_USE_OFFSET , 0x0000d0, 2, 0x0007ff00, 0x40008, 0x100, 0x0, 0, 0x8, 1 }, + { XPSB_RELOC_OP_OFFSET , 0x0000e8, 5, 0xffffffff, 0x0, 0x45800, 0x0, 0, 0x0, 0 }, + { XPSB_RELOC_OP_USE_REG , 0x0000ec, 2, 0x0000000f, 0x0, 0x140, 0x200000, 0, 0x10, 0 }, + { XPSB_RELOC_OP_USE_OFFSET , 0x0000ec, 2, 0x000000f0, 0xf0004, 0x140, 0x0, 0, 0x10, 0 }, + { XPSB_RELOC_OP_USE_OFFSET , 0x0000ec, 2, 0x0007ff00, 0x40008, 0x140, 0x0, 0, 0x10, 0 }, + { XPSB_RELOC_OP_OFFSET , 0x001120, 0, 0xffffffff, 0x0, 0x43e4, 0x0, 0, 0x0, 0 }, + { XPSB_RELOC_OP_USE_REG , 0x001124, 2, 0x0000000f, 0x0, 0x160, 0x200000, 0, 0x10, 0 }, + { XPSB_RELOC_OP_USE_OFFSET , 0x001124, 2, 0x000000f0, 0xf0004, 0x160, 0x0, 0, 0x10, 0 }, + { XPSB_RELOC_OP_USE_OFFSET , 0x001124, 2, 0x0007ff00, 0x40008, 0x160, 0x0, 0, 0x10, 0 }, + { XPSB_RELOC_OP_OFFSET , 0x001138, 0, 0xffffffff, 0x0, 0x44c4, 0x0, 0, 0x0, 0 }, + { XPSB_RELOC_OP_USE_REG , 0x00113a, 2, 0x0000000f, 0x0, 0x180, 0x200000, 0, 0x10, 0 }, + { XPSB_RELOC_OP_USE_OFFSET , 0x00113a, 2, 0x000000f0, 0xf0004, 0x180, 0x0, 0, 0x10, 0 }, + { XPSB_RELOC_OP_USE_OFFSET , 0x00113a, 2, 0x0007ff00, 0x40008, 0x180, 0x0, 0, 0x10, 0 }, + { XPSB_RELOC_OP_OFFSET , 0x000007, 0, 0x0fffffff, 0x40000, 0x44e0, 0x40000000, 5, 0x0, 0 }, + { XPSB_RELOC_OP_OFFSET , 0x00000a, 0, 0xffffffff, 0x0, 0x4464, 0x0, 5, 0x0, 0 }, + { XPSB_RELOC_OP_OFFSET , 0x00000c, 0, 0x0fffffff, 0x40000, 0x4480, 0x0, 5, 0x0, 0 }, + { XPSB_RELOC_OP_OFFSET , 0x001149, 0, 0x00ffffff, 0x40000, 0x380, 0x0, 0, 0x0, 0 }, + { XPSB_RELOC_OP_OFFSET , 0x00114b, 0, 0x00ffffff, 0x40000, 0x340, 0xc000000, 0, 0x0, 0 }, + { XPSB_RELOC_OP_OFFSET , 0x001158, 0, 0xffffffff, 0x0, 0x451c, 0x0, 0, 0x0, 0 }, + { XPSB_RELOC_OP_USE_REG , 0x00115a, 2, 0x0000000f, 0x0, 0x1a0, 0x200000, 0, 0x10, 0 }, + { XPSB_RELOC_OP_USE_OFFSET , 0x00115a, 2, 0x000000f0, 0xf0004, 0x1a0, 0x0, 0, 0x10, 0 }, + { XPSB_RELOC_OP_USE_OFFSET , 0x00115a, 2, 0x0007ff00, 0x40008, 0x1a0, 0x0, 0, 0x10, 0 }, + { XPSB_RELOC_OP_OFFSET , 0x00000e, 0, 0x0fffffff, 0x40000, 0x4560, 0x40000000, 5, 0x0, 0 }, + { XPSB_RELOC_OP_OFFSET , 0x000011, 0, 0xffffffff, 0x0, 0x3e4, 0x0, 5, 0x0, 0 }, + { XPSB_RELOC_OP_OFFSET , 0x000013, 0, 0x0fffffff, 0x40000, 0x3a0, 0x0, 5, 0x0, 0 }, + { XPSB_RELOC_OP_OFFSET , 0x000015, 0, 0x0fffffff, 0x40000, 0x180, 0x60000000, 5, 0x0, 0 }, +}; + +const struct xpsb_reloc xpsb_raster_relocs[XPSB_NUM_RAS_RELOCS] = { + /* op, where, buffer, mask, shift, pre_add, background, dst_buffer, arg0, arg1 */ + { XPSB_RELOC_OP_OFFSET , 0x010001, 1, 0xfffffffc, 0x20002, 0x0, 0x0, 0, 0x0, 0 }, + { XPSB_RELOC_OP_OFFSET , 0x01000a, 0, 0xffffffff, 0x0, 0x40000, 0x0, 0, 0x0, 0 }, + { XPSB_RELOC_OP_USE_REG , 0x01000c, 2, 0x0000000f, 0x0, 0x2000, 0x200000, 0, 0x10, 1 }, + { XPSB_RELOC_OP_USE_OFFSET , 0x01000c, 2, 0x000000f0, 0xf0004, 0x2000, 0x0, 0, 0x10, 1 }, + { XPSB_RELOC_OP_USE_OFFSET , 0x01000c, 2, 0x0007ff00, 0x40008, 0x2000, 0x0, 0, 0x10, 1 }, + { XPSB_RELOC_OP_USE_REG , 0x010008, 2, 0x0000000f, 0x0, 0x2020, 0x0, 0, 0x8, 1 }, + { XPSB_RELOC_OP_USE_OFFSET , 0x010008, 2, 0x000000f0, 0xf0004, 0x2020, 0x0, 0, 0x8, 1 }, + { XPSB_RELOC_OP_USE_OFFSET , 0x010008, 2, 0x0007ff00, 0x40008, 0x2020, 0x0, 0, 0x8, 1 }, + { XPSB_RELOC_OP_USE_REG , 0x010030, 2, 0x0000000f, 0x0, 0x2040, 0x100000, 0, 0x8, 1 }, + { XPSB_RELOC_OP_USE_OFFSET , 0x010030, 2, 0x000000f0, 0xf0004, 0x2040, 0x0, 0, 0x8, 1 }, + { XPSB_RELOC_OP_USE_OFFSET , 0x010030, 2, 0x0007ff00, 0x40008, 0x2040, 0x0, 0, 0x8, 1 }, + { XPSB_RELOC_OP_OFFSET , 0x01003a, 3, 0xffffffff, 0x0, 0x0, 0x0, 0, 0x0, 0 }, + { XPSB_RELOC_OP_OFFSET , 0x000400, 0, 0x00ffffff, 0x40000, 0x40100, 0x0, 4, 0x0, 0 }, + { XPSB_RELOC_OP_OFFSET , 0x000402, 0, 0x00ffffff, 0x40000, 0x400c0, 0xc000000, 4, 0x8, 1 }, + { XPSB_RELOC_OP_OFFSET , 0x008003, 4, 0x00ffffff, 0x40000, 0x1000, 0x2000000, 5, 0x8, 1 }, + { XPSB_RELOC_OP_OFFSET , 0x008017, 0, 0xffffffff, 0x40004, 0x40020, 0x0, 5, 0x8, 1 }, +}; + +/* Buffer 0 is the parameter heap. Every non-zero dword is written from this + * table rather than replayed as a byte blob, so the layout is visible and the + * size-dependent fields are computed rather than carried. + * + * Decoded and computed by xpsb_gen_heap(): the two render-target surface + * descriptors (0x000/0x003/0x04b), the tile-count shadow (0x05a), twelve size + * floats, the present rectangle and the viewport quad (state group 8). + * + * The remaining entries are constants whose bit-level meaning is not fully + * decoded. They are named by region rather than understood field by field - + * stated plainly rather than dressed up as generation. */ + + +/* Everything in the heap that describes the render target: the two surface + * descriptors, the tile shadow, the twelve size floats, the present rectangle + * and the viewport quad. sp is a stride in pixels. */ +static void heap_dest_fields(uint32_t *h, int w, int hgt, unsigned sp, + uint32_t fmt) +{ + /* The twelve size floats are the x and y of three full-screen quads - + * (x, y, z, w) records at heap+0x1bc, +0x43e4 and +0x4424, vertices + * (0,0), (w,0), (0,h), (w,h) - so six carry w and six carry h. */ + static const unsigned w_floats[6] = { + 0x073, 0x07b, 0x10fd, 0x1105, 0x110d, 0x1115 + }; + static const unsigned h_floats[6] = { + 0x078, 0x07c, 0x1102, 0x1106, 0x1112, 0x1116 + }; + float fw = (float)w, fh = (float)hgt; + float hw2 = fw * 0.5f, hh2 = fh * 0.5f, nh2 = -hh2; + unsigned i; + + h[0x000] = (sp - 1u) << 15; + h[0x003] = ((unsigned)(hgt - 1) << 12) | (unsigned)(w - 1); + h[0x04b] = XPSB_SURF_FMT(fmt) | + ((unsigned)(w - 1) << 12) | (unsigned)(hgt - 1); + h[0x05a] = ((((unsigned)w + 15u) / 16u) << 16) | + (((unsigned)hgt + 15u) / 16u); + for (i = 0; i < 6; i++) { + memcpy(&h[w_floats[i]], &fw, 4); + memcpy(&h[h_floats[i]], &fh, 4); + } + /* Present destination: origin, extent and surface size. The tile range + * that actually bounds the blit is in the raster stream. */ + h[0x10002] = ((unsigned)XPSB_PRESENT_Y << 12) | (unsigned)XPSB_PRESENT_X; + h[0x10003] = (((unsigned)(XPSB_PRESENT_Y + hgt) - 1u) << 12) | + ((unsigned)(XPSB_PRESENT_X + w) - 1u); + h[0x10033] = XPSB_SURF_FMT(fmt) | + ((unsigned)(w - 1) << 12) | (unsigned)(hgt - 1); + /* XPSB_VP_SCALE multiplies the group 8 viewport floats, to tell an MTE + * that applies them from one that ignores them: if the picture does + * not follow this, the group is not in circuit. */ + { + const char *vs = XPSB_ENVS("XPSB_VP_SCALE"); + float k = vs && *vs ? (float)atof(vs) : 1.0f; + float a = hw2 * k, b = hh2 * k, c = nh2 * k; + uint32_t *vp = h + xpsb_heap_state_off(h, XPSB_STATE_VIEWPORT); + + memcpy(&vp[0], &a, 4); + memcpy(&vp[1], &a, 4); + memcpy(&vp[2], &b, 4); + memcpy(&vp[3], &c, 4); + } +} + +void xpsb_gen_heap(uint32_t *h, int w, int hgt, uint32_t stride, int depth_test) +{ + unsigned sp = stride ? stride : (((unsigned)w + 31u) & ~31u); + + /* Synthesised rather than transcribed - see xpsb_heap.c. What is still + * a literal table lives there too, and test_heap gates the migration + * by diffing the result against the capture dword for dword. */ + struct xpsb_heap_cfg cfg; + + cfg.w = w; cfg.h = hgt; cfg.stride = sp; cfg.fmt = XPSB_FMT_8888; + xpsb_heap_synth(h, &cfg); + + /* The per-draw ISP word 0x1148 was captured as 0x01d00000: compare func + * [24:22] = ALWAYS with depth write-disable [20] set. A depth buffer is + * attached, so enabling the test is a compare function (3 = LEQUAL). */ + if (depth_test) { + uint32_t *a = h + xpsb_heap_state_off(h, XPSB_STATE_ISP_A); + + *a = (*a & ~(7u << 22) & ~(1u << 20)) | (3u << 22); + } +} + +uint32_t xpsb_heap_extent(const uint32_t *h) +{ + return h ? h[XPSB_HEAP_EXTENT] : 0u; +} + +/* Inclusive, so a rectangle of h pixels writes h-1. + * + * Word 0x003 is the height in [23:12] and the **stride** in [11:0]: the + * generator writes (hgt - 1) << 12 | (sp - 1), and 0x000 carries the stride + * again. It is not the extent - that is 0x04b, which holds the format with the + * width and the height. Writing the rectangle's width here put it in the + * stride, so a pass rendering a rectangle read the surface's rows at the + * rectangle's width: a 48 wide damage rectangle addressed a 1600 wide surface + * 48 pixels a row, and the picture sheared back to the top left corner, one + * pixel further every row. A whole-surface pass never showed it because there + * the width and the stride are the same number. + * + * The stride is kept and only the height is narrowed. What bounds the far edge + * sideways is the ISP's render box, which xpsb_ta_raster_set_range() sets from + * the same rectangle. */ +void xpsb_heap_set_extent(uint32_t *h, unsigned w, unsigned hgt) +{ + (void)w; + if (!h || !hgt) + return; + h[XPSB_HEAP_EXTENT] = ((hgt - 1u) << 12) | + (h[XPSB_HEAP_EXTENT] & 0xfffu); +} + +/* The enable is the whole of blending in the heap: captures with ONE/ONE, + * SRC_ALPHA and equation MIN differ by no dwords at all, because the factors + * and the equation are compiled into the pixel program. */ +void xpsb_heap_set_blend(uint32_t *h, int on) +{ + uint32_t *a = h + xpsb_heap_state_off(h, XPSB_STATE_ISP_A); + + if (on) + *a |= XPSB_ISP_BLEND; + else + *a &= ~XPSB_ISP_BLEND; +} + +/* The state block's layout. A group's dword is one for the mask plus the + * sizes of every present group below it, so putting a group in or taking one + * out moves everything above it and the descriptor's count - which is what + * xpsb_heap_set_cull() used to do for group 12 alone, done for the block. */ +static const unsigned char state_group_size[XPSB_STATE_NGROUPS] = { + 1, 1, 1, 1, 1, 1, /* ISP A B C, back-face A B C */ + 3, 2, 6, 1, 1, 1, 1, /* PDS, region clip, viewport, wrap, outsel, wclamp, MTE control */ + 0, 1, 1, /* terminate, texsize, texfloat */ +}; + +unsigned xpsb_state_size(unsigned group) +{ + return group < XPSB_STATE_NGROUPS ? state_group_size[group] : 0u; +} + +unsigned xpsb_state_dwords(uint32_t mask) +{ + unsigned g, n = 1; + + for (g = 0; g < XPSB_STATE_NGROUPS; g++) + if (mask & (1u << g)) + n += state_group_size[g]; + return n; +} + +/* Home while what the DMA carries fits between the mask and the descriptor, + * else the run past it. Seventeen dwords are carried as eighteen and go to + * the alternate: measured, a seventeen-dword block at home reaching the + * descriptor's dword renders nothing while eighteen at the alternate does. */ +static unsigned state_base_for(unsigned carried) +{ + return carried <= XPSB_STATE_HOME_ROOM ? XPSB_HEAP_STATE : + XPSB_HEAP_STATE_ALT; +} + +unsigned xpsb_dma_ctl_dwords(uint32_t ctl) +{ + return ((ctl & 0xfu) + 1u) * (((ctl >> 4) & 0xfu) + 1u); +} + +unsigned xpsb_heap_state_regs(const uint32_t *h) +{ + return (xpsb_dma_ctl_dwords(h[XPSB_HEAP_STATE_DESC + 1]) + 3u) / 4u; +} + +/* By the mask's presence, not by the descriptor's count: the count is the + * DMA's, which carries a dword more than a block of 17, 19 or 23. */ +unsigned xpsb_heap_state_base(const uint32_t *h) +{ + return h[XPSB_HEAP_STATE] ? XPSB_HEAP_STATE : XPSB_HEAP_STATE_ALT; +} + +uint32_t xpsb_state_dma_ctl(unsigned n, unsigned *carried) +{ + unsigned m; + + for (m = n; m <= XPSB_STATE_MAX_DWORDS + 1u; m++) { + uint32_t ctl = xpsb_dma_ctl(m, 0); + + if (ctl) { + if (carried) + *carried = m; + return ctl; + } + } + return 0; +} + +/* The copy program's two instruction forms, from the captured fifteen-dword + * one (xpsb_gen_usse() slot 0x1a0, "mov.skipinv.repeat15 o0, pa0" and + * "emitst.end.freep #0, #15"): the repeat count is USE1 [15:12], one less + * than the copies (EURASIA_USE1_RMSKCNT_SHIFT, sgxdefs.h:5207), the + * destination USE0 [27:21] and the source [13:7] (EURASIA_USE0_DST_SHIFT + * :5295, SRC1_SHIFT :5301), and the emit's count is its source 1 immediate + * in the same [13:7] (BuildEMIT, usecodegen.h:1458-1490). The captured + * slots at 0x80, 0xe0 and 0x180 - two, eleven and six - read the same way. */ +#define STATE_COPY_MOV_W1 0x28a10001u +#define STATE_COPY_MOV_W0 0xa0000000u +#define STATE_COPY_EMIT_W1 0xfb274000u +#define STATE_COPY_EMIT_W0 0xa0200000u + +unsigned xpsb_usse_state_copy(uint32_t *slot, unsigned n) +{ + unsigned at = 0, from = 0; + + if (!slot || !n || n > XPSB_USSE_STATE_MAX) + return 0; + while (from < n) { + unsigned k = n - from > 16u ? 16u : n - from; + + slot[at++] = STATE_COPY_MOV_W0 | (from << 21) | (from << 7); + slot[at++] = STATE_COPY_MOV_W1 | ((k - 1u) << 12); + from += k; + } + slot[at++] = STATE_COPY_EMIT_W0 | (n << 7); + slot[at++] = STATE_COPY_EMIT_W1; + return at * 4u; +} + +unsigned xpsb_usse_state_copy_off(unsigned n) +{ + return XPSB_USSE_STATE_OFF + (n - 1u) * XPSB_USSE_STATE_SLOT; +} + +unsigned xpsb_usse_state_copy_size(unsigned n) +{ + return n > 16u ? 24u : 16u; +} + +int xpsb_heap_state_off(const uint32_t *h, unsigned group) +{ + unsigned base = xpsb_heap_state_base(h); + uint32_t mask = h[base]; + unsigned g, off = 1; + + if (group >= XPSB_STATE_NGROUPS || !(mask & (1u << group))) + return -1; + for (g = 0; g < group; g++) + if (mask & (1u << g)) + off += state_group_size[g]; + return (int)(base + off); +} + +unsigned xpsb_heap_pds_dw(const uint32_t *h, unsigned w) +{ + int off = xpsb_heap_state_off(h, XPSB_STATE_PDS); + + /* The group is in every block this driver builds; a heap without it + * has no PDS binding to name and the captured dword is as good as any. */ + return (off < 0 ? XPSB_HEAP_STATE + 2u : (unsigned)off) + w; +} + +int xpsb_heap_state_set(uint32_t *h, unsigned group, int on, + const uint32_t *words) +{ + uint32_t blk[XPSB_STATE_MAX_DWORDS]; + unsigned base, nbase, n, nn, size, at, g, i, carried; + uint32_t mask, nmask, ctl; + + if (group >= XPSB_STATE_NGROUPS || (on && !words)) + return -EINVAL; + size = state_group_size[group]; + base = xpsb_heap_state_base(h); + mask = h[base]; + n = xpsb_state_dwords(mask); + if (mask & (1u << group)) { + if (on) { + memcpy(h + xpsb_heap_state_off(h, group), words, size * 4); + return 0; + } + nmask = mask & ~(1u << group); + } else { + if (!on) + return 0; + nmask = mask | (1u << group); + } + nn = xpsb_state_dwords(nmask); + ctl = xpsb_state_dma_ctl(nn, &carried); + if (!ctl) + return -ENOSPC; + nbase = state_base_for(carried); + if (carried > (nbase == XPSB_HEAP_STATE ? XPSB_STATE_HOME_ROOM : + XPSB_STATE_ALT_ROOM)) + return -ENOSPC; + /* Rebuilt group by group rather than shifted: the two blocks may not + * even overlap. */ + blk[0] = nmask; + at = 1; + for (g = 0; g < XPSB_STATE_NGROUPS; g++) { + if (!(nmask & (1u << g))) + continue; + if (g == group) + memcpy(blk + at, words, size * 4); + else + memcpy(blk + at, h + xpsb_heap_state_off(h, g), + state_group_size[g] * 4); + at += state_group_size[g]; + } + for (i = 0; i < n; i++) + h[base + i] = 0; + memcpy(h + nbase, blk, nn * 4); + h[XPSB_HEAP_STATE_DESC + 1] = ctl; + /* The descriptor's address follows the block. Its low byte is the + * block's offset in the 0x100-byte window, wherever the window is: the + * frame's at heap+0x4500, a record's copy at a multiple of 0x100. */ + h[XPSB_HEAP_STATE_DESC] = (h[XPSB_HEAP_STATE_DESC] & ~0xffu) | + ((nbase * 4u) & 0xffu); + return 0; +} + +int xpsb_heap_set_isp(uint32_t *h, const uint32_t *ff, const uint32_t *bf) +{ + int two = (ff[0] & XPSB_ISP_2SIDED) != 0; + int ret; + + if (two && !bf) + return -EINVAL; + /* Groups come out before any goes in, so the block never needs more + * room than the result does. */ + ret = xpsb_heap_state_set(h, XPSB_STATE_ISP_B, + (ff[0] & XPSB_ISP_BPRES) != 0, ff + 1); + if (!ret && !(ff[0] & XPSB_ISP_BPRES) && ff[0] & XPSB_ISP_CPRES) + ret = xpsb_heap_state_set(h, XPSB_STATE_ISP_C, 1, ff + 2); + if (!ret && !(ff[0] & XPSB_ISP_CPRES)) + ret = xpsb_heap_state_set(h, XPSB_STATE_ISP_C, 0, NULL); + if (!ret && !two) { + ret = xpsb_heap_state_set(h, XPSB_STATE_ISP_BF_A, 0, NULL); + if (!ret) + ret = xpsb_heap_state_set(h, XPSB_STATE_ISP_BF_B, 0, NULL); + if (!ret) + ret = xpsb_heap_state_set(h, XPSB_STATE_ISP_BF_C, 0, NULL); + } + if (!ret && (ff[0] & XPSB_ISP_BPRES) && ff[0] & XPSB_ISP_CPRES) + ret = xpsb_heap_state_set(h, XPSB_STATE_ISP_C, 1, ff + 2); + if (!ret && two) { + ret = xpsb_heap_state_set(h, XPSB_STATE_ISP_BF_A, 1, bf); + if (!ret) + ret = xpsb_heap_state_set(h, XPSB_STATE_ISP_BF_B, 1, bf + 1); + if (!ret) + ret = xpsb_heap_state_set(h, XPSB_STATE_ISP_BF_C, 1, bf + 2); + } + if (!ret) + ret = xpsb_heap_state_set(h, XPSB_STATE_ISP_A, 1, ff); + return ret; +} + +/* Group 12 is not in the captured block, so switching culling on is a layout + * change and not a store: the mask gains its bit, the word goes between group + * 11 and group 14, and the descriptor that DMAs the block has to name one more + * dword. Off restores the captured fifteen exactly, which is what keeps + * test_heap's dword-for-dword diff against the capture green. */ +void xpsb_heap_set_cull(uint32_t *h, uint32_t word, int on) +{ + xpsb_heap_state_set(h, XPSB_STATE_CULL, on, &word); +} + +/* DOUTI iterator control words, one per sampled unit, from the two captures + * that have this plumbing shape: tools/isa-pds/pds_corpus.txt P_mt2 for one + * unit and P_mt3 for two. Bits 3:0 are the texture coordinate set, and that is + * the only field this stream differs from the captures in - P_mt2 samples set + * 1 where the working stream samples set 0, and P_mt3 samples sets 1 and 2 + * where our vertex record carries 0 and 1. Everything else is verbatim: unit 0 + * of two drops the last-texture-issue bit and unit 1 carries no USE issue, + * because the iterated colour goes with the first unit's iterator. */ +static const uint32_t pds_itr[XPSB_NTEX_MAX][XPSB_NTEX_MAX] = { + { 0x0fc0aa00 }, + { 0x0fc0a200, 0x0c00fa01 }, +}; + +/* Placeholder texture addresses. The RC_TEX relocation replaces the whole + * dword, so only unit 0's is the value the captured heap carried. */ +static const uint32_t pds_tex_addr[XPSB_NTEX_MAX] = { 0x804ea000, 0x804ef000 }; + +static unsigned pri_pds_dwords(unsigned ntex) +{ + struct xpsb_pds_tex tex[XPSB_NTEX_MAX] = { { 0, 0, 0, 0 } }; + uint32_t use[3] = { 0 }, data[XPSB_PDS_MAX_DATA]; + uint32_t code[XPSB_PDS_MAX_CODE]; + unsigned nd = 0, nc, off; + + xpsb_pds_gen_primary(ntex, tex, use, data, &nd, code, &nc, &off); + return nd; +} + +/* Which coordinate set a set is to the part: a set carried on a colour + * iterator is not a coordinate set at all, so the ones that remain number + * from TC0 without the gaps they leave. */ +unsigned xpsb_attrib_tc_index(const struct xpsb_attribs *a, unsigned set) +{ + unsigned i, n = 0; + + for (i = 0; i < set && i < a->nset; i++) + if (!a->set[i].on_colour) + n++; + return n; +} + +/* The issue list an attribute configuration produces. The colour rides on the + * first issue and every set that is sampled or iterated gets one, in set + * order - which is also the order they claim primary attribute registers in. */ +static unsigned volume_texproj(void); + +static unsigned attrib_issue_list(const struct xpsb_attribs *a, + struct xpsb_pds_issue *out); + +/* The list, with the flat-shade corner stamped on whichever issue carries the + * colour - the packed one on USEISSUE_V0, or the coordinate set it rides in + * under FIX_HW_BRN_25211. The MTE's shade model is not enough on its own: the + * vendor writes the iterator's own FLATSHADE field from the same model + * (opengles1/usegles.c:2240-2252, validate.c:4476-4482), and gouraud there is + * a written value, not the reset one. + * + * Stamping it on an issue that also samples is legal - the field is the USE + * half's, and the vendor's generator merges a sampling issue onto a flat one + * without disturbing it (codegen/pds/pds.c:1176-1177). What is not supported + * is a TAG sampling the very set the program wants flat + * (usc2/regpack.c:5722-5726), which cannot arise here: a->colour_set is only + * ever a set nothing samples. */ +static unsigned attrib_issues(const struct xpsb_attribs *a, + struct xpsb_pds_issue *out) +{ + unsigned n = attrib_issue_list(a, out), i; + + if (!a->flatshade) + return n; + for (i = 0; i < n; i++) { + unsigned u = out[i].useissue; + + if (u == XPSB_DOUTI_COLOUR0 || + (a->colour_set && u == a->colour_set - 1u)) + out[i].flatshade = (uint8_t)a->flatshade; + } + return n; +} + +static unsigned attrib_issue_list(const struct xpsb_attribs *a, + struct xpsb_pds_issue *out) +{ + unsigned i, n = 0, j, carried = 0; + /* The frame carries XPSB_NSET_MAX sets and a->set[] holds that + * many, so a program reporting more would be read past the end of + * it and would name a set index the caller's per-set arrays do not + * have either. */ + unsigned nset = a->nset > XPSB_NSET_MAX ? XPSB_NSET_MAX : a->nset; + + /* An iterated set rides on a sampled set's issue rather than taking + * one of its own. The captured two-unit frame carries the iterated + * colour on unit 0's DOUTI - useissue 10 beside texissue 0 - and has + * two issues in all, not three (work/attrib-binding/README.md 3). + * Giving the colour an issue to itself made a third, and a DOUTI + * beyond what the frame expects does not deliver: section 4 measures + * the extra one as inert however it is placed. That is what left a + * second sampled unit reading state nothing ever wrote, and ioquake3 + * drew its lit surfaces black. */ + for (i = 0; i < nset; i++) { + /* Every unit reading the set, whether or not the set carries + * its coordinate: each needs its own DOUTT for the texture + * state SMP reads. */ + unsigned ntex = a->set[i].sampled > a->set[i].units ? + a->set[i].sampled : a->set[i].units; + unsigned u, nu = ntex ? ntex : 1u; + + if (!ntex && !a->set[i].iterated) + continue; + /* Iterated-only sets are placed by the sampled ones below, + * and only get an issue of their own if none is left to + * carry them. */ + if (!ntex) + continue; + /* One issue per unit reading the set - two units sharing a + * coordinate set is two issues on the same texissue, each + * with its own texture state and its own texel register. The + * set's iterated value rides on the first of them. */ + for (u = 0; u < nu && n < XPSB_PDS_MAX_ISSUE; u++) { + memset(&out[n], 0, sizeof out[n]); + out[n].useset = 0xffu; + /* SGX_TEXISSUE_ZERO points every texture issue at + * coordinate set zero. No capture has two texture + * issues naming different sets, so whether the field + * selects the set at all is a derivation; this tells + * a second unit that reads the wrong set from one + * that is handed no coordinate. + * + * Measured: no change. Pointed at set zero the two + * issues are word-identical to the one-set frame that + * works, and the second texel is still absent - so it + * is not the field. A recorded negative. */ + out[n].texissue = ntex ? + (XPSB_ENVS("SGX_TEXISSUE_ZERO") ? 0u : + (uint8_t)i) : XPSB_DOUTI_NONE; + /* Not a width: [9:8] of the word is the DDK's TEXPROJ, + * written here as the code plus one. The vendor's + * rule is one branch - a projected sample (TXP) asks + * for TEXPROJ_T and divides by the coordinate's own + * last component, and every other sample asks for + * TEXPROJ_RHW and is perspective-corrected against + * the RHW plane the record's fourth position float + * carries (opengles1/usegles.c:2209-2219, + * opengles2/use.c:2263-2273). TEXPROJ_S (3) is only + * on cores with SGX_FEATURE_CEM_S_USES_PROJ and is + * reserved here (sgxdefs.h:3695-3698). + * + * RHW is what lets a sampled set be two floats rather + * than three: under T the third float is the divisor + * and nothing else - a volume's slice was measured + * consumed by it, every coordinate selecting the last + * slice. "texdim 2 stalls the render" was this + * waiting on a plane the frame did not then declare; + * every frame declares one now. */ + out[n].texdim = a->set[i].projected ? 3 : + volume_texproj(); + out[n].useissue = XPSB_DOUTI_NONE; + out[n].usedim = 1; + { + unsigned it = nset; + + /* This set's own iterated value first, then + * the next set that is iterated and not + * sampled - the colour, in the shape the + * capture has. + * + * Every unit's issue may carry one, not only + * the first. A set read by two units takes an + * issue per unit and the second used to leave + * its DOUTI empty, so a program with three + * sets and two units on one of them wanted + * four issues where the data array holds + * three - the generator refused the list, the + * frame kept the template's iterators and + * every sample read the wrong coordinate. + * That is glmark2's jellyfish, whose two + * samplers share a set: filling the second + * issue's slot brings it to three and it + * renders. SGX_NO_CARRY_SPARE goes back to + * the first unit only. */ + if (!u && a->set[i].iterated) + it = i; + else if (u && XPSB_ENVS("SGX_NO_CARRY_SPARE")) + it = nset; + else + for (j = carried; j < nset; j++) + if (a->set[j].iterated && + !a->set[j].sampled && + !a->set[j].units) { + it = j; + carried = j + 1; + break; + } + if (it < nset) { + out[n].useissue = a->set[it].on_colour ? + (uint8_t)(a->set[it].on_colour == 2u ? + XPSB_DOUTI_COLOUR1 : XPSB_DOUTI_COLOUR0) : + (uint8_t)xpsb_attrib_tc_index(a, it); + out[n].usedim = a->set[it].width; + out[n].useset = (uint8_t)it; + out[n].unpacked = 1; + out[n].usef16 = a->set[it].f16; + } + } + n++; + } + } + /* Whatever no sampled set could carry - a program with no textures at + * all, or with more iterated sets than sampled ones - still needs an + * iterator, so it takes an issue of its own here. */ + for (i = carried; i < nset && n < XPSB_PDS_MAX_ISSUE; i++) { + if (!a->set[i].iterated || a->set[i].sampled || + a->set[i].units) + continue; + memset(&out[n], 0, sizeof out[n]); + out[n].useset = 0xffu; + out[n].texissue = XPSB_DOUTI_NONE; + out[n].texdim = 3; + out[n].useissue = a->set[i].on_colour ? + (uint8_t)(a->set[i].on_colour == 2u ? + XPSB_DOUTI_COLOUR1 : XPSB_DOUTI_COLOUR0) : + (uint8_t)xpsb_attrib_tc_index(a, i); + out[n].usedim = a->set[i].width; + out[n].useset = (uint8_t)i; + out[n].unpacked = 1; + out[n].usef16 = a->set[i].f16; + n++; + } + /* A frame that issues no texture at all renders many times slower than + * one that does, for the same geometry and the same pixels: an + * iterated set on its own costs 106 ms where the same scene beside a + * sampled set costs 5.7. Every captured frame issues a texture, and + * the shapes this driver runs fast are the ones that do. A program + * that samples for itself asks for no DOUTT, so give the first issue + * one anyway - the texel it fetches is simply not read. + * + * Measured on glmark2's textured cube, which is exactly this shape: + * 226 ms a frame to 17.4, render 106 ms to 6.6. The feature suite is + * unchanged at 58 of 63 with no faults, the textured cases among them, + * and the cube's brightness and coverage match the frame it drew + * before. SGX_NO_DUMMY_TEX puts it back. */ + /* A program with no coordinate set still gets a list of its own: the + * one the tail below would add to any other program, the TAG issue a + * textureless pixel task needs, carrying the packed colour where the + * program reads one. Without it the frame kept whatever list the + * previous program installed - its sets, its texture issue and its + * punch-through dependency (pds_use_for_issues) - against records and + * an ISP word that describe none of them, which stopped the core on a + * plain quad after a discarding one. */ + if (!nset && !XPSB_ENVS("SGX_NO_EMPTY_ATTRIBS")) { + memset(&out[0], 0, sizeof out[0]); + out[0].useset = 0xffu; + out[0].texissue = 0; + out[0].texdim = 1; + out[0].useissue = XPSB_DOUTI_NONE; + out[0].usedim = 1; + out[0].dummy = 1; + n = 1; + } + if (n && !XPSB_ENVS("SGX_NO_DUMMY_TEX")) { + unsigned q; + + for (q = 0; q < n; q++) + if (out[q].texissue != XPSB_DOUTI_NONE) + break; + if (q == n) { + unsigned pick = n; + + /* An issue of its own at the end, which displaces no + * set. On by default for the single-issue shape: the + * vendor driver never submits a frame without a + * texture issue, and the measured cost of one is the + * 18x - glmark2's untextured scenes at 2 frames a + * second against 53. The result is the captured + * two-issue shape exactly. The tail becomes texture + * unit zero; a record that binds no texture gets the + * context's own buffer patched in so its fetch lands + * on mapped memory. Marked so xpsb_heap_set_attribs() + * can drop it when the program does not fit, rather + * than losing the frame. SGX_NO_DUMMY_TAIL turns it + * off. + * + * A frame carrying two live sets gets one too, as a + * third issue. That used to be refused because where + * the part expects the texel register between two + * live sets was unestablished - but a tail issue does + * not sit between them, it follows both, and it is + * the convert form that moved a register and rendered + * ioquake3's world flat. The convert form stays + * behind SGX_DUMMY_ANY as the recorded experiment. + * + * Costly to refuse: glmark2's phong and cel iterate + * two sets and sample nothing, so they were the one + * shape that never got a texture issue, and a + * textureless pixel task without one dribbles - the + * same scene with the dummy taken away by + * SGX_NO_DUMMY_TAIL goes from 7 fps to 1. + * + * Safe by construction: if the third issue makes the + * PDS program outgrow the room before the vertex + * descriptor, the shed below drops the dummy first + * and the frame is exactly what it was. + * + * A tail only while the generator can build one. + * Three live sets already fill the program, so the + * fourth issue was appended here, refused there and + * shed - and the frame went out with no TAG issue at + * all. The fold below is what those frames get + * instead. */ + /* Against the spare block, not the captured slot: a + * list of four fits there and xpsb_pri_pds_off() + * moves the program to it on its own. Three + * coordinate sets and a dummy is the shape glmark2's + * jellyfish needs, and refusing it here left that + * scene with no texture issue at all. */ + if (xpsb_pds_list_fits(n + 1u, 1, + XPSB_PDS_MAX_DATA) && + !XPSB_ENVS("SGX_NO_DUMMY_TAIL")) { + memset(&out[n], 0, sizeof out[n]); + out[n].useset = 0xffu; + /* Set 0, unless asked otherwise. Every frame reads its + * first coordinate set through this issue, and a sampler + * takes a fixed width from a set - which is the shape of + * the group-14 rule, where TC0 must be a UV pair once a + * second set follows it. SGX_DUMMY_SET moves it. */ + out[n].texissue = 0; + if (XPSB_ENVS("SGX_DUMMY_SET")) + out[n].texissue = (uint8_t)atoi( + XPSB_ENVS("SGX_DUMMY_SET")); + /* No projection. This field is DOUTI[9:8], + * TEXPROJ, written as texdim - 1: a real + * sampled set asks for RHW so its fetch is + * perspective correct, and a three-component + * one asks for T. The dummy fetches a texel + * nothing reads, so a projection is at best + * a per-pixel divide for nothing - and it + * divides by a component of the coordinate + * set that a narrow varying never emits. */ + out[n].texdim = 1; + out[n].useissue = XPSB_DOUTI_NONE; + out[n].usedim = 1; + out[n].dummy = 1; + n++; + } else if (XPSB_ENVS("SGX_DUMMY_FOLD") && + out[n - 1].texissue == XPSB_DOUTI_NONE) { + /* No room for one of its own, so the TAG + * rides on the last issue. That is where the + * tail's texel register would have been - + * xpsb_attrib_bases() gives an issue its + * iterated registers and then its texel, so a + * fold onto the last issue moves no set's + * base, while the SGX_DUMMY_ANY form below + * folds onto an arbitrary one and moves every + * base behind it. The vendor merges the same + * way, copying EURASIA_DOUTI_TAG_MASK into + * the iterator before it + * (PDSGeneratePixelShaderProgram). + * + * Opt-in until it is measured. It has never + * run on the part - the run that was meant to + * test it carried none of these changes - and + * a three-set frame that renders nothing is + * worse than one that renders the wrong + * colour, so the default stays the shape the + * part has been running. SGX_DUMMY_FOLD=1. */ + out[n - 1].texissue = 0; + out[n - 1].texdim = 1; + out[n - 1].dummy = 1; + } else if (XPSB_ENVS("SGX_DUMMY_ANY")) { + for (q = 0; q < n; q++) { + unsigned u = out[q].useissue; + + if (u == XPSB_DOUTI_NONE || + u >= nset || !a->set[u].values) { + pick = q; + break; + } + } + if (pick >= n) + pick = 0; + out[pick].texissue = 0; + /* The same width a real sample of that set + * would ask for. Three told the TAG to read a + * third component from a set declared two + * floats wide, and it waited on one the + * iterator never sends: the core stalled with + * no MMU fault, which is plasma-welcome's + * lost frames. */ + /* The same width a real sample of that set + * would ask for. Three told the TAG to read a + * third component from a set declared two + * floats wide, and it waited on one the + * iterator never sends: the core stalled with + * no MMU fault, which is plasma-welcome's + * lost frames. */ + out[pick].texdim = (a->nset && + a->set[0].projected) ? + 3 : volume_texproj(); + out[pick].dummy = 1; + } + } + } + /* An issue that iterates and one that samples, merged into the one + * issue the part is ever given. Every captured frame's issues carry a + * texture - the two-unit word pair is 0fc0a200, 0c00fa01, the first of + * which samples unit 0 and iterates the colour on the same issue. + * Split across two, with the first carrying no texture at all, the + * sample returns nothing: the frame runs faultless and the texel is + * zero. One issue can name both, and the register layout is the same + * either way - the iterated floats then the texel. + * + * It does not help: merged, the texel is still zero. Kept behind + * SGX_ISSUE_MERGE as a recorded negative, since the shape it produces + * is the one the captured frames use and that is worth not retrying + * from scratch. + * + * Retried anyway in 2026-09 as the vendor's general rule rather than + * this two-issue case - PDSGeneratePixelShaderProgram() folds an + * iterator that carries a texture into the issue before it when that + * one carries none, copying only EURASIA_DOUTI_TAG_MASK: TEXISSUE, + * TEXCENTROID, TEXWRAP, TEXPROJ and TEXLASTISSUE. Written that way it + * is **worth nothing measurable** - glmark2's texture, shading and + * bump scenes come out at the same frame rate with it on and off, each + * run on its own - and it costs the fragment-discard case once a + * previous draw has left the part's transform armed. + * + * The apparent gain that made it look worth having (texture:linear 29 + * fps to 50) came from *this* block being enabled by the same + * variable at the same time, which is the negative above: it is fast + * because the texel it samples is zero. Two merges under one name. + * Anyone measuring here should set only one at a time and check the + * texel, not the frame rate. */ + if (n == 2 && out[0].texissue == XPSB_DOUTI_NONE && + out[0].useissue != XPSB_DOUTI_NONE && + out[1].texissue != XPSB_DOUTI_NONE && + out[1].useissue == XPSB_DOUTI_NONE && + XPSB_ENVS("SGX_ISSUE_MERGE")) { + out[0].texissue = out[1].texissue; + out[0].texdim = out[1].texdim; + n = 1; + } + /* The colour goes on the first issue when that issue iterates nothing + * of its own, and on one of its own otherwise. Without any set at all + * it is the whole list - the vendor's untextured word. + * + * But not when a set is iterated. The loop above has already built + * that set's word - texissue none, useissue naming the set, unpacked, + * usedim its width - which is what xpsb_pds.h's decode of the + * vendor's primary emitter calls a varying. The insert below is + * conditional and these three assignments were not, so the varying was + * pushed to issue 1 and a packed colour nobody asked for took issue 0, + * where the compiled program looks for its input. It read an 8888 + * dword as floats - denormals near 1e-43 - and glmark2's gouraud + * rendered its whole canvas black. */ + for (i = 0; i < nset; i++) + if (a->set[i].iterated) + break; + /* An iterated set already holds issue 0, so a colour the program reads + * takes one of its own at the end - in front, it moved the set every + * read is based on. With no issue at all its base stayed 0, which is + * set 0's, and alacritty read its background out of the coordinates. */ + if (i < nset) { + if (!a->colour || n >= XPSB_PDS_MAX_ISSUE) + return n; /* no room: the sets still stand */ + memset(&out[n], 0, sizeof out[n]); + out[n].useset = 0xffu; + out[n].texissue = XPSB_DOUTI_NONE; + out[n].useissue = XPSB_DOUTI_COLOUR0; + out[n].usedim = 4; + out[n].unpacked = 0; + return n + 1; + } + /* And when issue 0 iterates nothing of its own it carries the colour + * rather than needing one inserted - which is what the comment above + * describes and what the captured two-unit frame does, useissue ten + * beside its texissue. Nothing reached that case: the insert below + * fires only when there is no issue at all or when issue 0's useissue + * is already taken, so a lone sampled set - whose useissue is none - + * left the colour with no issue and no register, and + * xpsb_attrib_bases() then left its base at zero, which is that set's + * own. The program unpacked the coordinate as a colour for it. */ + if (n && a->colour && out[0].useissue == XPSB_DOUTI_NONE) { + out[0].useissue = XPSB_DOUTI_COLOUR0; + out[0].usedim = 4; + out[0].unpacked = 0; + return n; + } + if (!n || out[0].useissue != XPSB_DOUTI_NONE) { + if (n >= XPSB_PDS_MAX_ISSUE) + return 0; + memmove(out + 1, out, n * sizeof *out); + memset(&out[0], 0, sizeof out[0]); + out[0].useset = 0xffu; + out[0].texissue = XPSB_DOUTI_NONE; + out[0].usedim = 4; + n++; + } + out[0].useissue = XPSB_DOUTI_COLOUR0; + out[0].usedim = 4; + out[0].unpacked = 0; + return n; +} + +unsigned xpsb_attrib_issues(const struct xpsb_attribs *a, unsigned *nissue, + unsigned char *issue_of) +{ + struct xpsb_pds_issue issue[XPSB_PDS_MAX_ISSUE]; + unsigned n = attrib_issues(a, issue), i, u = 0; + + for (i = 0; i < n; i++) + if (issue[i].texissue != XPSB_DOUTI_NONE) { + if (issue_of && u < XPSB_MAX_TEX) + issue_of[u] = (unsigned char)i; + u++; + } + if (nissue) + *nissue = n; + return u; +} + +/* Where the coordinate sets begin in the record: behind the position quad + * and the colour quads the MTE emits. The base colour quad is always there; + * the offset colour one only when a set is carried on V1. */ +unsigned xpsb_attrib_set_base(const struct xpsb_attribs *a) +{ + unsigned i, n = 8; + + for (i = 0; i < a->nset; i++) + if (a->set[i].on_colour == 2u) + return 12u; + return n; +} + +unsigned xpsb_attrib_stride(const struct xpsb_attribs *a) +{ + unsigned i, n = xpsb_attrib_set_base(a); + + if (XPSB_ENVS("SGX_SET_BASE")) + n = (unsigned)atoi(XPSB_ENVS("SGX_SET_BASE")); + + for (i = 0; i < a->nset; i++) + if (!a->set[i].on_colour) + n += a->set[i].width; + /* Experiment: trailing floats no set names, so the record widens + * without the frame iterating any more components. */ + if (XPSB_ENVS("SGX_STRIDE_PAD")) + n += (unsigned)atoi(XPSB_ENVS("SGX_STRIDE_PAD")); + return n; +} + +void xpsb_attrib_bases(const struct xpsb_attribs *a, unsigned *colour, + unsigned *set_base, unsigned *tex_base) +{ + struct xpsb_pds_issue issue[XPSB_PDS_MAX_ISSUE]; + unsigned n = attrib_issues(a, issue), i, reg = 0, s = 0; + + if (colour) + *colour = 0; + if (XPSB_ENVS("SGX_DUMP_ISSUES")) { + unsigned q; + + fprintf(stderr, "xpsb: in nset %u colour %u", a->nset, + a->colour); + for (q = 0; q < a->nset; q++) + fprintf(stderr, " [w %u samp %u iter %u f16 %u]", + a->set[q].width, a->set[q].sampled, + a->set[q].iterated, a->set[q].f16); + fprintf(stderr, "\n"); + fprintf(stderr, "xpsb: %u issue(s):", n); + for (q = 0; q < n; q++) + fprintf(stderr, " [tex %d dim %u, use %d dim %u " + "unpacked %u f16 %u]", + (int)(int8_t)issue[q].texissue, + issue[q].texdim, (int)(int8_t)issue[q].useissue, + issue[q].usedim, issue[q].unpacked, + issue[q].usef16); + fprintf(stderr, "\n"); + } + for (i = 0; i < n; i++) { + if (issue[i].useissue != XPSB_DOUTI_NONE) { + if (issue[i].unpacked) { + /* The caller's array holds one base per set, + * so an issue naming anything else - a colour + * on useissue ten, or a set index past the two + * the frame carries - would write past the end + * of it. It is a stack array in both callers, + * which is a corrupted return address rather + * than a wrong picture. The texture bases below + * have carried this guard since three samplers + * on one varying did exactly that. */ + if (set_base && issue[i].useset < XPSB_NSET_MAX) + set_base[issue[i].useset] = reg; + reg += xpsb_douti_regs(issue[i].usedim, + issue[i].usef16); + } else { + if (colour) + *colour = reg; + reg++; + } + } + if (issue[i].texissue != XPSB_DOUTI_NONE) { + /* One entry per texture issue, not per set: several + * samplers can read one coordinate set - three on one + * varying is the ordinary shape of a lit surface - and + * capping this at the set count silently dropped every + * base past the second, so those units read a register + * the iterator never filled and sampled black. The + * bound is the issue list's own. */ + if (tex_base && s < XPSB_MAX_TEX) + tex_base[s] = reg; + s++; + reg++; + } + } + (void)i; +} + +/* Rebuilds the primary PDS program in place. The USE task words and the + * texture state words are read back out of the heap rather than carried here, + * so whatever xpsb_frame_set_texture() wrote survives and the ntex == 1 form + * regenerates the captured dwords exactly. */ +static void heap_set_pri_pds(uint32_t *h, unsigned ntex) +{ + struct xpsb_pds_tex tex[XPSB_NTEX_MAX]; + uint32_t data[XPSB_PDS_MAX_DATA], code[XPSB_PDS_MAX_CODE], use[3]; + uint32_t *ctl, *pri; + unsigned nd, nc, off, u; + + use[0] = h[XPSB_PRI_PDS_DW + xpsb_pds_ds_dword(0, 0)]; + use[1] = h[XPSB_PRI_PDS_DW + xpsb_pds_ds_dword(0, 1)]; + use[2] = h[XPSB_PRI_PDS_DW + xpsb_pds_ds_dword(1, 0)]; + for (u = 0; u < ntex; u++) { + tex[u].ctl = h[xpsb_texctl_dw_at(XPSB_PRI_PDS_OFF, u)]; + tex[u].fmt = h[xpsb_texstate_dw_at(XPSB_PRI_PDS_OFF, u)]; + tex[u].addr = pds_tex_addr[u]; + tex[u].itr = pds_itr[ntex - 1][u]; + } + if (xpsb_pds_gen_primary(ntex, tex, use, data, &nd, code, &nc, &off)) + return; + + /* Only the region this base owns: the arrays are sized for the widest + * list the spare block can hold, and clearing that much from the + * captured slot ran over the vertex descriptor behind it. */ + memset(h + XPSB_PRI_PDS_DW, 0, + XPSB_PRI_PDS_END - XPSB_PRI_PDS_OFF); + memcpy(h + XPSB_PRI_PDS_DW, data, nd * 4); + memcpy(h + XPSB_PRI_PDS_DW + off / 4, code, nc * 4); + /* the bare secondary program shares heap+0x380, which the code segment + * only reaches at two units */ + if (XPSB_PRI_PDS_OFF + off + nc * 4 <= XPSB_SEC_PDS_NULL) + h[XPSB_SEC_PDS_NULL / 4] = XPSB_PDS_HALT; + + ctl = h + xpsb_heap_pds_dw(h, XPSB_PDS_W_CTL); + pri = h + xpsb_heap_pds_dw(h, XPSB_PDS_W_PRI); + *ctl = (*ctl & ~XPSB_VARYING_MASK) | ((1u + ntex) & XPSB_VARYING_MASK); + *pri = (*pri & 0x00ffffffu) | (nd << 24); +} + +/* The vertex-fetch DMA descriptor's control word counts dwords, and the stride + * the fetch program multiplies the index by is in bytes. Both are derived from + * the record the caller writes, so they cannot drift from it. */ +/* A DOUTD control word for d dwords into attribute offset ao. + * + * The burst-size field is four bits, so the single-line XPSB_DMA_CTL() wraps + * silently past sixteen dwords. This is the divisor search the stock driver + * runs at psb_dri.so:0x381c5 - the largest burst below sixteen that divides + * the transfer - giving a contiguous multi-line descriptor instead. */ +uint32_t xpsb_dma_ctl(unsigned d, unsigned ao) +{ + unsigned bs = d, i; + + if (!d) + return 0; + if (d > 16) { + for (i = 15; i > 1; i--) + if (d % i == 0 && d / i <= 16) + break; + if (i <= 1) + return 0; + bs = i; + } + return 0x80000000u | ((bs - 1u) << 21) | (((d / bs) - 1u) << 4) | + (bs - 1u) | (ao << 8); +} + +void xpsb_heap_set_ntex(uint32_t *h, unsigned ntex) +{ + unsigned nfloat = xpsb_vtx_stride(ntex); + uint32_t ctl = xpsb_dma_ctl(nfloat, 0); + + if (ntex < 1 || ntex > XPSB_NTEX_MAX || !ctl) return; + h[XPSB_VTXDMA_DW] = ctl; + h[XPSB_VTXSTRIDE_DW] = nfloat * 4u; + heap_set_pri_pds(h, ntex); +} + +/* The TEXPROJ an unprojected sample's issue carries, as texdim (the code plus + * one): RHW, the vendor's answer for every sample that is not a TXP; + * SGX_3D_TEXPROJ=<0..3> puts any of the four codes in for the experiment. */ +static unsigned volume_texproj(void) +{ + const char *e = XPSB_ENVS("SGX_3D_TEXPROJ"); + + if (e && *e >= '0' && *e <= '3') + return (unsigned)(*e - '0') + 1u; + return 2u; +} + +unsigned xpsb_attrib_texproj(const struct xpsb_attribs *a, unsigned unit) +{ + struct xpsb_pds_issue issue[XPSB_PDS_MAX_ISSUE]; + unsigned n = attrib_issues(a, issue), i, u = 0; + + for (i = 0; i < n; i++) { + if (issue[i].texissue == XPSB_DOUTI_NONE) + continue; + if (u == unit) + return issue[i].texdim ? issue[i].texdim - 1u : 0u; + u++; + } + return 0; +} + +/* Group 14 holds three bits per coordinate set: 1 for two floats, 3 for three, + * 5 for three delivered projected - which is what a sampled set is - and 7 for + * four. The captured word is 0x00000005, one projected set. The DDK names + * them EURASIA_MTE_TEXDIM_UV, UVS, UVT and UVST (sgxdefs.h:2061-2065): the + * three-float form is S present, so it is also a volume's slice coordinate. */ +/* One set's three-bit code, which is also what says where its projection + * divisor is: T is the last component the code names. */ +static unsigned tcset_code(const struct xpsb_attribs *a, unsigned i) +{ + /* By the set's own component count, which is the whole of the + * vendor's rule: two floats are UV, three UVS and four UVST, and the + * vertex size grows by that many (opengles2/shader.c:2245-2270, + * opengles1/fftnlgles.c:1024-1090). Nothing there declares a UVT set + * for an ordinary sample - T is the projection divisor, and a sample + * that needs one is a projected sample, which carries the coordinate + * and the divisor together. + * + * This used to answer 5 - UVT - for every sampled set, so a two-float + * coordinate cost three record floats and the third was read as a + * divisor. attrib_issues() asks for TEXPROJ_RHW instead and the + * divisor comes off the position's RHW plane. */ + unsigned w = a->set[i].width & 7u; + + if (a->set[i].projected) + return 7; + if (w >= 4) + return 7; + if (w == 3) + return 3; + return 1; +} + +/* Which float of a set the TAG divides the coordinate by, or -1 for a set it + * does not project. + * + * TEXPROJ_T divides by T, and the declared dimension says which float that + * is: UVT names three components and T is the third, UVST names four and T is + * the fourth (EURASIA_MTE_TEXDIM_*, sgxdefs.h:2061-2065). So it is always the + * set's last float, and a record that writes the divisor at a fixed third one + * hands a four-float set a divisor of whatever the varying's fourth component + * held - zero, for a coordinate pair Mesa packed into one varying, which + * leaves the unit sampling one texel for every pixel. */ +int xpsb_attrib_proj_float(const struct xpsb_attribs *a, unsigned set) +{ + unsigned code; + + if (!a || set >= a->nset || !a->set[set].sampled) + return -1; + /* Only a projected sample is divided by a component of its own set; + * every other one is TEXPROJ_RHW and divided by the position's plane, + * so no float of the set is a divisor. */ + if (!a->set[set].projected) + return -1; + code = tcset_code(a, set); + if (!(code & 4u)) /* no T, so nothing is divided by it */ + return -1; + return (int)(a->set[set].width & 7u) - 1; +} + +/* Group 14 holds three bits per coordinate set: 1 for two floats, 3 for three, + * 5 for three delivered projected - which is what a sampled set is - and 7 for + * four. The captured word is 0x00000005, one projected set. The DDK names + * them EURASIA_MTE_TEXDIM_UV, UVS, UVT and UVST (sgxdefs.h:2061-2065): the + * three-float form is S present, so it is also a volume's slice coordinate. */ +static uint32_t attrib_tcset_word(const struct xpsb_attribs *a) +{ + /* SGX_TCSET_WORD overrides group 14 outright. The codes below past + * 5 and 7 are extrapolation - no capture has a second three-bit + * field set at all - so a sweep is the only way to tell whether this + * word is what a two-set frame gets wrong. */ + { + const char *e = XPSB_ENVS("SGX_TCSET_WORD"); + + if (e && *e) + return (uint32_t)strtoul(e, NULL, 0); + } + + uint32_t w = 0; + unsigned i; + + for (i = 0; i < a->nset; i++) + if (!a->set[i].on_colour) + w |= tcset_code(a, i) << + (3 * xpsb_attrib_tc_index(a, i)); + return w; +} + +/* The record's width, in the three places that describe it: the vertex DMA's + * dword count, the byte stride the fetch walks and the group-10 field the MTE + * reads. They have to name one width, and a caller that writes its vertices + * at a wider record than its attributes need says so here. */ +int xpsb_heap_set_record_width(uint32_t *h, unsigned nfloat, + int offset_colour) +{ + uint32_t ctl = xpsb_dma_ctl(nfloat, 0); + uint32_t *g10; + + if (!h || !ctl) { + errno = EINVAL; + return -1; + } + h[XPSB_VTXDMA_DW] = ctl; + h[XPSB_VTXSTRIDE_DW] = nfloat * 4u; + g10 = h + xpsb_heap_state_off(h, XPSB_STATE_OUTSEL); + *g10 = (*g10 & ~XPSB_STATE_G10_DW) | + ((nfloat << XPSB_STATE_G10_SHIFT) & XPSB_STATE_G10_DW); + if (offset_colour) + *g10 |= XPSB_STATE_G10_OFFSET; + else + *g10 &= ~XPSB_STATE_G10_OFFSET; + return 0; +} + + +/* The pixel data-master control word. + * + * This is CalculatePixelDMSInfo() from the DDK (eurasia/common/dmscalc), for + * SGX535's constants, and it exists because the three task-size fields are not + * independent of the shader: they are all derived from how many primary + * attribute registers a pixel costs. The driver used to write only the pixel + * size and leave PDSTASKSIZE and USETASKSIZE at whatever the captured frame + * had - 24 blocks and 4 - which is a budget computed for the capture's own + * pixel size. + * + * A PDS task may claim at most a quarter of the primary attribute space + * (PVR_ATTRIBUTES_DM_SUBDIVISION). The capture's 24 blocks is already above + * that for a two-component varying and survives; at three or four components + * the per-pixel size rounds up from 4 registers to 8, the same 24 blocks now + * asks for twice the space, and the task can never be allocated. The pixel + * task then never launches, the render makes no progress at all, and the + * kernel times it out as a locked core - which is what glmark2's build and + * shading scenes did on every frame. + * + * temps of zero means "not known here"; the task size is then bounded only by + * the attribute space, which is what the capture's own value assumes. */ +#define XPSB_DMS_CHUNK 32u /* EURASIA_PDS_CHUNK_SIZE */ +#define XPSB_DMS_BLOCK_PIXELS 4u /* PVR_BLOCK_SIZE_IN_PIXELS */ +#define XPSB_DMS_SEC_SUBDIV 3u /* ..._SUBDIVISION_FOR_PIXELS */ +#define XPSB_DMS_DM_SUBDIV 4u /* PVR_ATTRIBUTES_DM_SUBDIVISION */ +#define XPSB_DMS_MAX_VS_SEC 300u /* PVR_MAX_VS_SECONDARIES */ +#define XPSB_DMS_USETASK_MAX 4u /* EURASIA_PDS_USETASKSIZE_MAXSIZE + 1 */ +#define XPSB_DMS_PDSTASK_MAX 32u /* EURASIA_PDS_PDSTASKSIZE_MAX */ +#define XPSB_DMS_USE_PDS_RATIO 8u /* ..._USESIZETOPDSSIZE_MAX_RATIO */ +/* The attribute register budget, as sgxinit.c derives it: + * EURASIA_USE_NUM_UNIFIED_REGISTERS less the output partitions and the + * microkernel's own secondary attributes. The first two are headers; the + * third is the microkernel's, and one chunk is the smallest it can be, so + * this is the largest budget the arithmetic may assume. */ +#define XPSB_DMS_UNIFIED_REGS 2048u /* EURASIA_USE_NUM_UNIFIED_REGISTERS */ +/* The whole temporary region, EURASIA_USE_GLOBAL_TEMP_REG_LIMIT: sgxinit.c + * programs EUR_CR_USE_TMPREG from it and the part reads back 0x44, which is + * that arithmetic exactly - granularity 4, 24 initialised, 384 registers. */ +#define XPSB_DMS_TEMP_TOTAL 384u /* EURASIA_USE_GLOBAL_TEMP_REG_LIMIT */ +#define XPSB_DMS_OUT_PARTITION 64u /* EURASIA_OUTPUT_PARTITION_SIZE */ +#define XPSB_DMS_OUT_PARTITIONS 6u /* EURASIA_USSE_NUM_OUTPUT_PARTITIONS */ +#define XPSB_DMS_UKERNEL_SA XPSB_DMS_CHUNK +/* The temporaries come out of the same store and sit below the attributes: + * sgxinit.c derives ui32AttribRegStart from ui32NumUSETemporaryRegisters, so + * the attribute budget starts above the whole temporary region. Leaving that + * out made this budget 384 registers too generous, which is the input the + * note below suspected. */ +#define XPSB_DMS_ATTR_REGS (XPSB_DMS_UNIFIED_REGS - \ + XPSB_DMS_TEMP_TOTAL - \ + XPSB_DMS_OUT_PARTITION * \ + XPSB_DMS_OUT_PARTITIONS - \ + XPSB_DMS_UKERNEL_SA) +/* The share of the store one pixel task's temporaries may use: + * PVR_TEMPORARIES_SUBDIVISION of the temporary region, not of the unified + * store - dmscalc.c divides psSGXInfo->ui32NumUSETemporaryRegisters, which + * sgxinit.c reports as the 384 EUR_CR_USE_TMPREG was programmed from. Taken + * from the unified store this was 1024, and a task was let hold four blocks + * of a 24-temporary program where the vendor allows two. */ +#define XPSB_DMS_TEMP_SUBDIV 2u /* PVR_TEMPORARIES_SUBDIVISION */ +#define XPSB_DMS_TEMP_REGS (XPSB_DMS_TEMP_TOTAL / XPSB_DMS_TEMP_SUBDIV) + +static unsigned align_up(unsigned v, unsigned a) +{ + return a ? ((v + a - 1u) / a) * a : v; +} + +uint32_t xpsb_pds_dms_word(uint32_t cur, unsigned pixel_regs, unsigned temps, + unsigned sa_dwords) +{ + unsigned pa_px, attr_chunks, sa_chunks, pa_chunks; + unsigned use_task, use_chunks, pds_chunks, pds_blocks, quarter; + + /* Rounded to four for the USSE's stride granularity, which is what + * makes five registers cost the same as eight. */ + pa_px = align_up(pixel_regs, 4u); + if (!pa_px) + return cur; + + sa_chunks = (sa_dwords + XPSB_DMS_CHUNK - 1u) / XPSB_DMS_CHUNK; + sa_chunks += (XPSB_DMS_MAX_VS_SEC + XPSB_DMS_CHUNK - 1u) / + XPSB_DMS_CHUNK; + attr_chunks = XPSB_DMS_ATTR_REGS / XPSB_DMS_CHUNK; + if (attr_chunks <= sa_chunks) + return cur; + pa_chunks = attr_chunks - sa_chunks; + + /* How many 2x2 blocks one USE task may hold, bounded by the temporary + * registers it needs and then by the attribute space of one + * subdivision. */ + use_task = XPSB_DMS_USETASK_MAX; + if (temps) { + unsigned tchunk = align_up(temps * XPSB_DMS_BLOCK_PIXELS, + XPSB_DMS_CHUNK); + + use_task = tchunk && XPSB_DMS_TEMP_REGS >= tchunk ? + XPSB_DMS_TEMP_REGS / tchunk : 1u; + if (use_task > XPSB_DMS_USETASK_MAX) + use_task = XPSB_DMS_USETASK_MAX; + } + use_chunks = (pa_chunks + XPSB_DMS_SEC_SUBDIV - 1u) / + XPSB_DMS_SEC_SUBDIV; + if (use_task * XPSB_DMS_BLOCK_PIXELS * pa_px > + use_chunks * XPSB_DMS_CHUNK) + use_task = (use_chunks * XPSB_DMS_CHUNK) / + (XPSB_DMS_BLOCK_PIXELS * pa_px); + if (!use_task) + use_task = 1u; + + /* And how many a PDS task may hold: the maximum-sized USE tasks that + * fit, plus the one smaller task at the end. */ + use_chunks = align_up(use_task * XPSB_DMS_BLOCK_PIXELS * pa_px, + XPSB_DMS_CHUNK) / XPSB_DMS_CHUNK; + if (!use_chunks) + use_chunks = 1u; + pds_chunks = (XPSB_DMS_SEC_SUBDIV - 1u) * (use_chunks - 1u); + pds_chunks = pa_chunks > pds_chunks ? pa_chunks - pds_chunks : 1u; + pds_blocks = (pds_chunks / use_chunks) * use_task + + ((pds_chunks % use_chunks) * XPSB_DMS_CHUNK) / + (XPSB_DMS_BLOCK_PIXELS * pa_px); + + /* The quarter of the attribute space a PDS task may not exceed. This + * is the clamp the captured word violates once a pixel costs eight + * registers rather than four. */ + quarter = (pa_chunks * XPSB_DMS_CHUNK) / XPSB_DMS_DM_SUBDIV; + if (pds_blocks * XPSB_DMS_BLOCK_PIXELS * pa_px > quarter) + pds_blocks = quarter > XPSB_DMS_BLOCK_PIXELS * pa_px ? + quarter / (XPSB_DMS_BLOCK_PIXELS * pa_px) : 1u; + if (pds_blocks > XPSB_DMS_PDSTASK_MAX) + pds_blocks = XPSB_DMS_PDSTASK_MAX; + if (pds_blocks > XPSB_DMS_USE_PDS_RATIO * use_task) + pds_blocks = XPSB_DMS_USE_PDS_RATIO * use_task; + if (!pds_blocks) + pds_blocks = 1u; + + cur &= ~(XPSB_DMS_PIXELSIZE_MASK | XPSB_DMS_USETASK_MASK | + XPSB_DMS_PDSTASK_MASK); + cur |= (pixel_regs & 0x7fu); + cur |= ((use_task - 1u) & 3u) << XPSB_DMS_USETASK_SHIFT; + cur |= (pds_blocks & 0x1fu) << XPSB_DMS_PDSTASK_SHIFT; + return cur; +} + +/* The task control words an issue list implies: what the task must wait for, + * and whether it is punch-through. Shared by the frame's own primary program + * and by the per-record ones, which have to agree with their own issue lists + * rather than inherit the frame's. */ +static void pds_use_for_issues(uint32_t use[3], unsigned n, + const struct xpsb_pds_issue *issue, + const struct xpsb_attribs *a) +{ + unsigned i; + + for (i = 0; i < n; i++) + if (issue[i].texissue != XPSB_DOUTI_NONE) + break; + if (i == n) + use[0] &= ~XPSB_DOUTU_TEXDEP; + else + use[0] |= XPSB_DOUTU_TEXDEP; + use[1] &= ~XPSB_DOUTU1_PUNCHTHROUGH; + if (a->uses_kill && !XPSB_ENVS("SGX_NO_PUNCHTHROUGH")) + use[1] |= XPSB_DOUTU1_PUNCHTHROUGH_PHASE1; + for (i = 0; i < n; i++) + if (issue[i].useissue != XPSB_DOUTI_NONE) + break; + if (i == n) + use[0] &= ~XPSB_DOUTU_ITERDEP; + else + use[0] |= XPSB_DOUTU_ITERDEP; +} + +/* The captured slot holds 24 dwords of data and code together, which is three + * sampled units - a fourth issue's address lands on memory dword 24. A list + * that does not fit is built in the free block above the video constants + * instead, which no relocation and nothing else in the heap names. Everything + * that fits stays exactly where the capture had it, so a frame that works + * today is byte-identical. */ +/* Drop the dummy texture issues so a list that overruns its region can be + * retried: one that also iterates keeps the iteration and loses the TAG, one + * that does nothing else comes off. Returns the count left, or 0 when there + * was nothing to shed. */ +static unsigned pds_shed_dummies(struct xpsb_pds_issue *issue, unsigned n, + uint32_t *use) +{ + unsigned i, kept = 0; + + for (i = 0; i < n; i++) { + if (!issue[i].dummy) { + issue[kept++] = issue[i]; + } else if (issue[i].useissue != XPSB_DOUTI_NONE) { + issue[i].texissue = XPSB_DOUTI_NONE; + issue[i].dummy = 0; + issue[kept++] = issue[i]; + } + } + if (kept == n || !kept) + return 0; + use[0] &= ~XPSB_DOUTU_TEXDEP; + for (i = 0; i < kept; i++) + if (issue[i].texissue != XPSB_DOUTI_NONE) + use[0] |= XPSB_DOUTU_TEXDEP; + return kept; +} + +static int pds_fits_captured(unsigned n, const struct xpsb_pds_issue *issue, + const uint32_t *use) +{ + uint32_t data[XPSB_PDS_MAX_DATA], code[XPSB_PDS_MAX_CODE]; + unsigned nd = 0, nc = 0, off = 0; + + return !xpsb_pds_gen_primary_issues_max(n, issue, use, data, &nd, code, + &nc, &off, + XPSB_PDS_CAP_DATA) && + XPSB_PRI_PDS_OFF + off + nc * 4 <= XPSB_PRI_PDS_END; +} + +/* As above, for the spare block: it takes a wider list, and + xpsb_heap_set_attribs() builds against XPSB_PDS_MAX_DATA there. */ +static int pds_fits_alt(unsigned n, const struct xpsb_pds_issue *issue, + const uint32_t *use) +{ + uint32_t data[XPSB_PDS_MAX_DATA], code[XPSB_PDS_MAX_CODE]; + unsigned nd = 0, nc = 0, off = 0; + + return !xpsb_pds_gen_primary_issues_max(n, issue, use, data, &nd, code, + &nc, &off, + XPSB_PDS_MAX_DATA) && + XPSB_PRI_PDS_ALT + off + nc * 4 <= XPSB_PRI_PDS_ALT_END; +} + +unsigned xpsb_pri_pds_off(const struct xpsb_attribs *a) +{ + struct xpsb_pds_issue issue[XPSB_PDS_MAX_ISSUE]; + uint32_t use[3] = { 0, 0, 0 }; + unsigned n, k; + + if (!a) + return XPSB_PRI_PDS_OFF; + n = attrib_issues(a, issue); + if (!n) + return XPSB_PRI_PDS_OFF; + /* What the chooser saw, which is the first thing to ask when a program + * with several sampled units stalls: how many issues its attributes + * came to, how many of those are texture issues, and which block would + * hold them. */ + if (XPSB_ENVS("SGX_PDS_ROOM")) { + unsigned q, ntex = 0; + + for (q = 0; q < n; q++) + if (issue[q].texissue != XPSB_DOUTI_NONE) + ntex++; + fprintf(stderr, "xpsb: pds: %u issue(s), %u texture, " + "captured %d, alt %d; nset %u", n, ntex, + pds_fits_captured(n, issue, use), + pds_fits_alt(n, issue, use), a->nset); + for (q = 0; q < a->nset; q++) + fprintf(stderr, " [w %u samp %u iter %u]", + a->set[q].width, a->set[q].sampled, + a->set[q].iterated); + fputc('\n', stderr); + } + /* The captured slot first, and shedding into it exactly as before, so + * every frame that renders today is laid out where it always was. The + * spare block is for the lists that have nowhere else to go. */ + if (pds_fits_captured(n, issue, use)) + return XPSB_PRI_PDS_OFF; + /* Before shedding: what the shed takes away first is the dummy + * texture issue, and a pixel task without one runs an order of + * magnitude slower. Three coordinate sets and a dummy is glmark2 + * jellyfish, and it is the shape that has nowhere else to go. */ + if (pds_fits_alt(n, issue, use)) + return XPSB_PRI_PDS_ALT; + k = pds_shed_dummies(issue, n, use); + if (k && pds_fits_captured(k, issue, use)) + return XPSB_PRI_PDS_OFF; + return XPSB_PRI_PDS_ALT; +} + +int xpsb_heap_set_attribs(uint32_t *h, const struct xpsb_attribs *a, + unsigned *pri_dwords) +{ + struct xpsb_pds_issue issue[XPSB_PDS_MAX_ISSUE]; + uint32_t data[XPSB_PDS_MAX_DATA], code[XPSB_PDS_MAX_CODE], use[3]; + uint32_t *ctl, *pri; + unsigned nd, nc, off, n, i, t = 0, nfloat; + unsigned base, base_dw, base_end; + + if (!h || !a || a->nset > XPSB_NSET_MAX) + return -1; + base = xpsb_pri_pds_off(a); + base_dw = base / 4; + base_end = base == XPSB_PRI_PDS_OFF ? XPSB_PRI_PDS_END : + XPSB_PRI_PDS_ALT_END; + /* A set carries two to four floats. Anything else indexes the record + * layout out of its table and encodes a group-14 width the iterator + * does not have, so it is refused here rather than rendered from. */ + for (i = 0; i < a->nset; i++) + if ((a->set[i].width & 7u) < 2u || (a->set[i].width & 7u) > 4u) + return -1; + n = attrib_issues(a, issue); + if (!n) + return -1; + /* Experiment: every issue's iterated width, leaving the record's + * stride as it is. Applied here because the list is built by several + * paths and one of them carries a set from another unit. */ + if (XPSB_ENVS("SGX_USEDIM")) { + unsigned q, wv = (unsigned)atoi(XPSB_ENVS("SGX_USEDIM")); + + for (q = 0; q < n; q++) + if (issue[q].useissue != XPSB_DOUTI_NONE) + issue[q].usedim = (uint8_t)wv; + } + + use[0] = h[XPSB_PRI_PDS_DW + xpsb_pds_ds_dword(0, 0)]; + use[1] = h[XPSB_PRI_PDS_DW + xpsb_pds_ds_dword(0, 1)]; + use[2] = h[XPSB_PRI_PDS_DW + xpsb_pds_ds_dword(1, 0)]; + /* The task's dependency bits have to match the issues it is actually + * given, and a discarding program is punch-through - the same rules a + * per-record program follows, so they live in one place. */ + pds_use_for_issues(use, n, issue, a); + for (i = 0; i < n; i++) { + unsigned u; + + if (issue[i].texissue == XPSB_DOUTI_NONE) + continue; + /* The heap carries the two units the capture had, so a third + * issue's words are read from the same unit its address comes + * from. Indexing past them took whatever followed the unit + * array as a texture's control and format words, and the + * fetch then ran past the bound buffer: an MMU fault inside + * the texture window with the cache as the requestor. */ + /* Always from the captured slot: that is where the driver + * writes a unit's words, whichever base the program is built + * at, so the two never have to agree on where they are. */ + u = t < XPSB_NTEX_MAX ? t : 0; + issue[i].ctl = h[xpsb_texctl_dw_at(xpsb_tex_unit_base(u), u)]; + issue[i].fmt = h[xpsb_texstate_dw_at(xpsb_tex_unit_base(u), u)]; + issue[i].addr = pds_tex_addr[u]; + t++; + } + nd = nc = off = 0; + /* The code segment must stay clear of the vertex PDS program's data + * at heap+0x3a0, which is what caps the issue list. Bounded against + * that offset and not against XPSB_VTXDMA_DW * 4, which is one dword + * past it and let the code overwrite the descriptor's first word. + * + * A program that does not fit sheds its dummy texture issue first: the + * dummy is a speed lever, not a correctness one, and failing here + * loses the whole frame. A converted issue goes back to iterating + * only; a tail issue comes off the list. A list the generator refuses + * outright - four issues overrun its data array - is the same case. */ + if (XPSB_ENVS("SGX_PDS_LIST")) { + unsigned q; + + fprintf(stderr, "xpsb: pds: %u issue(s):\n", n); + for (q = 0; q < n; q++) + fprintf(stderr, "xpsb: pds: %u: texissue %d texdim %u " + "useissue %d usedim %u unpacked %u dummy %d " + "word %08x\n", + q, (int)(int8_t)issue[q].texissue, + (unsigned)issue[q].texdim, + (int)(int8_t)issue[q].useissue, + (unsigned)issue[q].usedim, + (unsigned)issue[q].unpacked, + (int)issue[q].dummy, issue[q].word); + } + if (xpsb_pds_gen_primary_issues_max(n, issue, use, data, &nd, code, + &nc, &off, + base == XPSB_PRI_PDS_OFF ? + XPSB_PDS_CAP_DATA : + XPSB_PDS_MAX_DATA) || + base + off + nc * 4 > base_end) { + unsigned kept = 0; + + if (XPSB_ENVS("SGX_PDS_ROOM")) { + unsigned q; + + fprintf(stderr, "xpsb: pds: %u issue(s) want data %u + " + "code %u = %u byte(s) at 0x%x, and the vertex " + "descriptor is at 0x%x\n", n, off, nc * 4, + off + nc * 4, XPSB_PRI_PDS_OFF, + XPSB_PDS_VTXDESC_OFF); + /* Data and code of zero mean the generator refused the + * list rather than sizing it, so the count is what did + * not fit and the list itself is what to look at. */ + for (q = 0; q < n; q++) + fprintf(stderr, "xpsb: pds: issue %u: " + "texissue %d, useissue %d, dummy %d\n", + q, (int)issue[q].texissue, + (int)issue[q].useissue, + (int)issue[q].dummy); + } + + kept = pds_shed_dummies(issue, n, use); + if (!kept) + return -1; + n = kept; + if (xpsb_pds_gen_primary_issues_max(n, issue, use, data, &nd, + code, &nc, &off, + base == XPSB_PRI_PDS_OFF ? + XPSB_PDS_CAP_DATA : + XPSB_PDS_MAX_DATA)) + return -1; + if (base + off + nc * 4 > base_end) + return -1; + } + if (XPSB_ENVS("SGX_ISSUE_PROBE")) { + unsigned q, ntex = 0, ndum = 0; + + for (q = 0; q < n; q++) { + if (issue[q].texissue != XPSB_DOUTI_NONE) ntex++; + if (issue[q].dummy) ndum++; + } + fprintf(stderr, "sgx: issues %u, texture %u, dummy %u\n", + n, ntex, ndum); + } + + /* Everything this frame claims about the region, checked against the + * region itself before the part is asked to run it. Each of these was + * a real defect that reached the hardware as a core stall with no MMU + * fault, which says nothing about which number was wrong. */ + if (!XPSB_ENVS("SGX_NO_PDS_CHECK")) { + unsigned end = base + off + nc * 4; + unsigned q, hi, samples = 0; + + for (q = 0; q < n; q++) + if (issue[q].texissue != XPSB_DOUTI_NONE) + samples = 1; + hi = xpsb_pds_list_high_dw(n, (int)samples); + + if (base + hi * 4u >= base_end) + fprintf(stderr, "xpsb: pds: issue %u reaches dword %u " + "(0x%x), past the region ending 0x%x\n", + n - 1, hi, base + hi * 4u, base_end); + if (end > base_end) + fprintf(stderr, "xpsb: pds: code ends at 0x%x, past " + "the region ending 0x%x\n", end, base_end); + /* The captured slot has the secondary program's HALT behind + * it and the vertex fetch behind that; a program that reaches + * either leaves the frame running one of them as PDS code. */ + if (base == XPSB_PRI_PDS_OFF && end > XPSB_PDS_VTXDESC_OFF) + fprintf(stderr, "xpsb: pds: code ends at 0x%x, over " + "the vertex descriptor at 0x%x\n", end, + XPSB_PDS_VTXDESC_OFF); + /* A TAG told to read more components than the set it names is + * declared to carry waits for one the iterator never sends. */ + for (q = 0; q < n; q++) { + unsigned set, w; + + if (issue[q].texissue == XPSB_DOUTI_NONE || + !issue[q].texdim) + continue; + set = issue[q].texissue; + if (set >= a->nset) + continue; + w = a->set[set].width & 7u; + if (issue[q].texdim > w) + fprintf(stderr, "xpsb: pds: issue %u samples " + "set %u for %u component(s), and the " + "set carries %u\n", q, set, + issue[q].texdim, w); + } + } + memset(h + base_dw, 0, base_end - base); + memcpy(h + base_dw, data, nd * 4); + memcpy(h + base_dw + off / 4, code, nc * 4); + if (XPSB_PRI_PDS_OFF + off + nc * 4 <= XPSB_SEC_PDS_NULL) + h[XPSB_SEC_PDS_NULL / 4] = XPSB_PDS_HALT; + if (XPSB_ENVS("SGX_PDS_WROTE")) + fprintf(stderr, "xpsb: pds wrote: %u issue(s), data %u dword(s)" + ", code %u at 0x%x, first code word %08x, 0x380 now " + "%08x\n", n, nd, nc, XPSB_PRI_PDS_OFF + off, code[0], + h[XPSB_SEC_PDS_NULL / 4]); + + /* The whole word, not only the pixel size: the task sizes beside it + * are derived from the same per-pixel cost and the capture's 24 and 4 + * are a budget computed for the capture's own. See + * xpsb_pds_dms_word(). + * + * This used to be behind SGX_DMS_CALC, on the grounds that the + * fragment-discard case failed with it - which was + * attribs.ntemps being read before codegen had run, fixed in the same + * commit that added the calculation. A program with constants has had + * the computed word ever since, because the driver's secondary-PDS + * builder writes it again with its own secondary size; a program with + * none never entered that builder and so kept the capture's task + * sizes. Both get the arithmetic now, and the secondary size is the + * builder's to add. SGX_DMS_KEEP inherits the capture's for an A/B. */ + ctl = h + xpsb_heap_pds_dw(h, XPSB_PDS_W_CTL); + if (!XPSB_ENVS("SGX_DMS_KEEP")) + *ctl = xpsb_pds_dms_word(*ctl, xpsb_pds_issue_regs(n, issue), + a->ntemps, 0u); + else + *ctl = (*ctl & ~XPSB_VARYING_MASK) | + (xpsb_pds_issue_regs(n, issue) & XPSB_VARYING_MASK); + /* Experiment: the pixel size alone, leaving the record stride and the + * issues' usedim as they are, to tell the two apart. */ + if (XPSB_ENVS("SGX_PIXELSIZE")) + *ctl = (*ctl & ~XPSB_VARYING_MASK) | + ((unsigned)atoi(XPSB_ENVS("SGX_PIXELSIZE")) & + XPSB_VARYING_MASK); + if (XPSB_ENVS("SGX_DMS_PROBE")) + fprintf(stderr, "sgx: dms %08x: pixelsize %u, usetask %u, " + "pdstask %u, secattr %u (issue regs %u)\n", + *ctl, *ctl & 0x7fu, (*ctl >> 16) & 3u, + (*ctl >> 25) & 0x1fu, (*ctl >> 18) & 0x7fu, + xpsb_pds_issue_regs(n, issue)); + pri = h + xpsb_heap_pds_dw(h, XPSB_PDS_W_PRI); + *pri = (*pri & 0x00ffffffu) | (nd << 24); + if (pri_dwords) + *pri_dwords = nd; + + nfloat = xpsb_attrib_stride(a); + if (xpsb_heap_set_record_width(h, nfloat, + xpsb_attrib_set_base(a) > 8u)) + return -1; + h[xpsb_heap_state_off(h, XPSB_STATE_TEXSIZE)] = attrib_tcset_word(a); + /* The TA's per-set coordinate float format, beside the widths that + * name them. The vendor emits the two together + * (opengles2/shader.c:SetupProgramOutputSelects, its + * ui32TexCoordPrecision); a frame that emits only the widths runs on + * whatever that register last held. Every set this driver writes is + * 32-bit, so the word is zero - but it has to be said, not assumed. */ + if (!XPSB_ENVS("SGX_NO_TEXFLOAT")) { + uint32_t tf = 0; + const char *e = XPSB_ENVS("SGX_TEXFLOAT_WORD"); + + if (e && *e) + tf = (uint32_t)strtoul(e, NULL, 0); + xpsb_heap_state_set(h, XPSB_STATE_TEXFLOAT, 1, &tf); + } + return 0; +} + +/* The record the vertex DMA fetched, emitted by this program. The repeat + * count is four bits, so one mov carries sixteen dwords and no more: a record + * of twenty had 19 masked to 3 and emitted four, which is the position and + * nothing else, and every varying past it reached the iterator as whatever + * the register last held. That is glmark2's terrain drawn in one flat colour + * - it needs twenty - and it is why a record wider than sixteen "does not + * draw". Wide records take a mov apiece, the same shape + * xpsb_usse_state_copy() already uses, and the template's own emit follows + * them. The slot is thirty-two bytes, which holds three movs and the emit. */ +#define XPSB_VTX_COPY_MOV_W0 0xa0000000u +#define XPSB_VTX_COPY_MOV_W1 0x28a10001u +#define XPSB_VTX_COPY_EMIT_W0 0xa0200000u +#define XPSB_VTX_COPY_EMIT_W1 0xfb275000u + +void xpsb_usse_set_vtx_dwords(uint32_t *u, unsigned dwords) +{ + uint32_t *w = u + XPSB_USSE_VTX_OFF / 4; + unsigned at = 0, from = 0; + + if (!dwords) + return; + if (dwords <= 16u || XPSB_ENVS("SGX_NO_WIDE_VTX")) { + w[1] = (w[1] & ~XPSB_USSE_VTX_MASK) | + (((dwords - 1u) << 12) & XPSB_USSE_VTX_MASK); + return; + } + /* Three movs and the emit is the whole slot; a record past that would + * run into the next program, so it keeps the narrow form and the + * caller's own width check refuses it. */ + if (dwords > 48u) + return; + while (from < dwords) { + unsigned k = dwords - from > 16u ? 16u : dwords - from; + + w[at++] = XPSB_VTX_COPY_MOV_W0 | (from << 21) | (from << 7); + w[at++] = XPSB_VTX_COPY_MOV_W1 | ((k - 1u) << 12); + from += k; + } + w[at++] = XPSB_VTX_COPY_EMIT_W0; + w[at++] = XPSB_VTX_COPY_EMIT_W1; +} + +/* ---- caller-supplied surfaces ---- */ + +/* The row pitch of a linear (TEXTYPE_STRIDE) surface, in bytes. + * + * The descriptor carries no pitch - xpsb_tex_state_mode() writes a width and + * a height and nothing else - so the unit derives one, and the layout, the + * upload and the sampler check all have to agree with whatever it derives. + * The vendor's own figure for this core is ALIGN(width, 32) texels: + * sgxfeaturedefs.h:189-224 gives SGX535 SGX_FEATURE_TEXTURE_STRIDE_ + * GRANULARITY_32, sgxdefs.h:5015-5044 turns that into EURASIA_TAG_STRIDE_ + * ALIGN0 = ALIGN1 = 32 with THRESHOLD 0, and opengles1/texmgmt.c:1873-1886 + * compares a surface's own stride against exactly that before deciding it + * needs an explicit one. (The granularity-8 figure in the same header is + * SGX520's, one block up, and reading it as ours cost a wrong theory once.) + * + * The suite says the derived pitch is nevertheless not what a one-byte + * surface is read at: lumrows gets row 0 of a 2x2 GL_LUMINANCE right and row + * 1 black, while texquad reads row 1 of a 2x2 four-byte texture correctly at + * this same rule. So the rule is a parameter until one sweep settles it - + * SGX_TEX_PITCH_ALIGN sets the texel granularity (32 by default) and + * SGX_TEX_PITCH_MIN a floor in bytes (none by default). Both are read here, + * where the layout, the view and both surface checks reach them, so the + * three cannot disagree. */ +unsigned xpsb_linear_stride(uint32_t format, uint32_t width) +{ + unsigned bpp = xpsb_format_bpp(format); + unsigned align = 32u, min = 0u, n; + const char *e; + + if (!bpp || !width) + return 0; + e = XPSB_ENVS("SGX_TEX_PITCH_ALIGN"); + if (e && *e) { + unsigned v = (unsigned)strtoul(e, NULL, 0); + + /* A power of two, or the pitch is not a pitch. */ + if (v && !(v & (v - 1u))) + align = v; + } + e = XPSB_ENVS("SGX_TEX_PITCH_MIN"); + if (e && *e) + min = (unsigned)strtoul(e, NULL, 0); + n = bpp * ((width + align - 1u) & ~(align - 1u)); + return n < min ? min : n; +} + +unsigned xpsb_format_bpp(uint32_t format) +{ + switch (format) { + case XPSB_FMT_A8: return 1; + case XPSB_FMT_AL88: + case XPSB_FMT_4444: + case XPSB_FMT_1555: + case XPSB_FMT_565: + case XPSB_FMT_YUY2: + case XPSB_FMT_UYVY: return 2; + case XPSB_FMT_8888: + case XPSB_FMT_BGR8888: return 4; + case XPSB_FMT_U16: + case XPSB_FMT_S16: + case XPSB_FMT_F16: return 2; + case XPSB_FMT_U1616: + case XPSB_FMT_S1616: + case XPSB_FMT_F1616: + case XPSB_FMT_F32: return 4; + default: return 0; + } +} + +/* There is no pitch field anywhere in the texture state, so the hardware reads + * a surface at exactly this stride. A caller whose surface differs has to be + * refused rather than sampled wrongly. */ +static int exact_log2(uint32_t v) +{ + int n = 0; + + if (!v || (v & (v - 1u))) + return -1; + while (v > 1u) { v >>= 1; n++; } + return n; +} + +int xpsb_surface_check(const struct xpsb_surface_desc *s) +{ + unsigned bpp = xpsb_format_bpp(s->format); + int lw, lh; + + /* A block format has no pitch and no texel size; it is described by + * its log2 sizes alone, which is the twiddled form. */ + if (s->format == XPSB_FMT_ETC1 && !s->twiddled) { + errno = EINVAL; + return -1; + } + if ((!bpp && s->format != XPSB_FMT_ETC1) || !s->w || !s->h || + s->w > 4096 || s->h > 4096) { + errno = EINVAL; + return -1; + } + /* A twiddled surface is addressed by log2 of its dimensions and holds + * no pitch at all, so the stride rule below does not apply to it and + * a non-power-of-two size has no encoding. */ + if (s->twiddled) { + lw = exact_log2(s->w); + lh = exact_log2(s->h); + if (lw < 0 || lh < 0) { + errno = EINVAL; + return -1; + } + } else if (s->stride != xpsb_linear_stride(s->format, s->w)) { + errno = EINVAL; + return -1; + } + /* A volume is a log2 encoding in all three axes - SSIZE is four bits + * like USIZE and VSIZE - so it is twiddled or it is nothing. */ + if (s->depth > 1 && + (!s->twiddled || s->depth > 4096 || exact_log2(s->depth) < 0)) { + errno = EINVAL; + return -1; + } + if (s->nlevels > 1) { + uint32_t d = s->w > s->h ? s->w : s->h, n = 1; + + if (s->depth > d) + d = s->depth; + + while (d > 1u) { d >>= 1; n++; } + if (s->nlevels > n || s->nlevels > 15u) { + errno = EINVAL; + return -1; + } + } + if (s->mipfilter > 1 || s->twiddled > 1) { + errno = EINVAL; + return -1; + } + return 0; +} + +uint32_t xpsb_twiddle_index(uint32_t x, uint32_t y, + uint32_t log2w, uint32_t log2h) +{ + uint32_t n = log2w < log2h ? log2w : log2h; + uint32_t idx = 0, c; + + for (c = 0; c < n; c++) { + idx |= ((y >> c) & 1u) << (2u * c); + idx |= ((x >> c) & 1u) << (2u * c + 1u); + } + /* Past the square part only one axis still has bits, and they sit + * above the interleaved block in their original order. */ + if (log2w > n) + idx |= (x >> n) << (2u * n); + else if (log2h > n) + idx |= (y >> n) << (2u * n); + return idx; +} + +uint32_t xpsb_twiddle_level_offset(uint32_t w, uint32_t h, uint32_t level) +{ + uint32_t off = 0, i; + + for (i = 0; i < level; i++) { + off += w * h; + w = w > 1u ? w >> 1 : 1u; + h = h > 1u ? h >> 1 : 1u; + } + return off; +} + +static int twiddle_both(void *tw, void *lin, uint32_t w, uint32_t h, + uint32_t lin_stride, unsigned bpp, int to_twiddled) +{ + int lw = exact_log2(w), lh = exact_log2(h); + unsigned char *t8 = tw, *l8 = lin; + uint32_t x, y; + + if (lw < 0 || lh < 0 || !bpp || !tw || !lin) { + errno = EINVAL; + return -1; + } + for (y = 0; y < h; y++) { + for (x = 0; x < w; x++) { + uint32_t i = xpsb_twiddle_index(x, y, + (uint32_t)lw, + (uint32_t)lh); + unsigned char *a = t8 + (size_t)i * bpp; + unsigned char *b = l8 + (size_t)y * lin_stride + + (size_t)x * bpp; + + if (to_twiddled) memcpy(a, b, bpp); + else memcpy(b, a, bpp); + } + } + return 0; +} + +int xpsb_twiddle_level(void *dst, const void *src, uint32_t w, uint32_t h, + uint32_t src_stride, unsigned bpp) +{ + return twiddle_both(dst, (void *)(uintptr_t)src, w, h, src_stride, + bpp, 1); +} + +int xpsb_untwiddle_level(void *dst, const void *src, uint32_t w, uint32_t h, + uint32_t dst_stride, unsigned bpp) +{ + return twiddle_both((void *)(uintptr_t)src, dst, w, h, dst_stride, + bpp, 0); +} + +uint32_t xpsb_twiddle3_index(uint32_t x, uint32_t y, uint32_t z, + uint32_t log2w, uint32_t log2h, uint32_t log2d, + enum xpsb_twiddle3 order) +{ + uint32_t idx = 0, bit = 0, c; + + if (order == XPSB_TWIDDLE3_SLICES) + return (z << (log2w + log2h)) | + xpsb_twiddle_index(x, y, log2w, log2h); + /* One bit from each axis that still has one, y then x then z, until + * only one axis is left - whose remaining bits are then contiguous, + * which is what the 2D rule does with the longer axis. */ + for (c = 0; c < log2w || c < log2h || c < log2d; c++) { + if (c < log2h) + idx |= ((y >> c) & 1u) << bit++; + if (c < log2w) + idx |= ((x >> c) & 1u) << bit++; + if (c < log2d) + idx |= ((z >> c) & 1u) << bit++; + } + return idx; +} + +uint32_t xpsb_twiddle3_level_offset(uint32_t w, uint32_t h, uint32_t d, + uint32_t level) +{ + uint32_t off = 0, i; + + for (i = 0; i < level; i++) { + off += w * h * d; + w = w > 1u ? w >> 1 : 1u; + h = h > 1u ? h >> 1 : 1u; + d = d > 1u ? d >> 1 : 1u; + } + return off; +} + +static int twiddle3_both(void *tw, void *lin, uint32_t w, uint32_t h, + uint32_t d, uint32_t lin_stride, uint32_t lin_slice, + unsigned bpp, enum xpsb_twiddle3 order, + int to_twiddled) +{ + int lw = exact_log2(w), lh = exact_log2(h), ld = exact_log2(d); + unsigned char *t8 = tw, *l8 = lin; + uint32_t x, y, z; + + if (lw < 0 || lh < 0 || ld < 0 || !bpp || !tw || !lin) { + errno = EINVAL; + return -1; + } + for (z = 0; z < d; z++) + for (y = 0; y < h; y++) + for (x = 0; x < w; x++) { + uint32_t i = xpsb_twiddle3_index(x, y, z, + (uint32_t)lw, (uint32_t)lh, + (uint32_t)ld, order); + unsigned char *a = t8 + (size_t)i * bpp; + unsigned char *b = l8 + (size_t)z * lin_slice + + (size_t)y * lin_stride + + (size_t)x * bpp; + + if (to_twiddled) memcpy(a, b, bpp); + else memcpy(b, a, bpp); + } + return 0; +} + +int xpsb_twiddle3_level(void *dst, const void *src, uint32_t w, uint32_t h, + uint32_t d, uint32_t src_stride, uint32_t src_slice, + unsigned bpp, enum xpsb_twiddle3 order) +{ + return twiddle3_both(dst, (void *)(uintptr_t)src, w, h, d, src_stride, + src_slice, bpp, order, 1); +} + +int xpsb_untwiddle3_level(void *dst, const void *src, uint32_t w, uint32_t h, + uint32_t d, uint32_t dst_stride, uint32_t dst_slice, + unsigned bpp, enum xpsb_twiddle3 order) +{ + return twiddle3_both((void *)(uintptr_t)src, dst, w, h, d, dst_stride, + dst_slice, bpp, order, 0); +} + +/* EURASIA_TAG_BORDERMAP_OFFSET_x (sgxdefs.h:5064-5075, the !SGX545 + * branch) is "offset to first map face, in texels" and runs 56, 64, 80, ... + * 16432 for 1x1 to 2048x2048. The steps are 8, 16, 32, ... - 8n between the + * entry for n and the entry for 2n - so the table is one prefix sum: the + * corner block first (CORNERS_OFFSET 52, padded to 56, sgxdefs.h:5061), then + * the face for each size in turn, the face for size n occupying 8n texels. + * 545's table is the same form four texels further along, its corner block + * being 56 (sgxdefs.h:5080-5096), and it confirms the reading. + * + * This is the offset the unit itself uses. Handed an address with a border + * address mode selected, it fetches at that address plus this many texels - + * measured on the part at 2x2 (256 bytes), 4x4 (320) and 8x8 (448), the last + * of them to the byte. So the address in DOUTT2 is the map's base and this is + * where the texture's own texels begin. */ +uint32_t xpsb_border_map_face_offset(uint32_t size) +{ + /* EURASIA_TEXTURESIZE_MAX is 2048 on this core (sgxdefs.h:423) and + * the table's last entry is 2048x2048; a larger texture has no + * border map at all rather than an extrapolated one. */ + if (!size || size > 2048u || exact_log2(size) < 0) + return 0; + return 48u + 8u * size; +} + +/* The table's whole extent for that size: everything up to its own face, and + * that face too, 8n texels further on. + * + * This is NOT what a texture puts in front of its texels. Measured 2026-09-03 + * at 2x2, 4x4 and 8x8: the unit fetches at the address it is handed plus + * xpsb_border_map_face_offset(size), exactly, so level 0 belongs at the face + * offset and not past the whole extent. The function stays because the extent + * is what the DDK's table describes and the host suite checks the prefix sum + * against it; xpsb_border_map_face_offset() is the one the layout uses. */ +uint32_t xpsb_border_map_texels(uint32_t size) +{ + uint32_t face = xpsb_border_map_face_offset(size); + + return face ? face + 8u * size : 0u; +} + +/* Word 1 agrees with both decodes bit for bit. Word 0 differs: a clamp/nearest + * 8888 texture is 0x03fe0090 in the captured GL stream and 0x001e0092 in + * Xpsb.so. + * + * That is no longer a mystery. Reversing psb_dri.so identified word 0 bits + * 26:21 as the LOD bias field, value (code - 31) / 8, named by its own assert + * string "lodbiasi <= 0x3f" - so 31 is exactly bias 0. The two words decode as + * bias +0.000 for the capture and bias -3.875, the minimum, for Xpsb. Xpsb + * gets away with it because its textures are never mipmapped and the LOD + * clamps to 0 regardless. + * + * So the capture's value is the neutral one and stays the default, and the + * only genuinely unexplained bit between the two is bit 1. Both forms occur in + * the working frame - heap+0x348 is the capture's, heap+0x4a and heap+0x10032 + * are Xpsb's - so neither is untried on this silicon. */ +int xpsb_tex_state_mode(uint32_t *w0, uint32_t *w1, + const struct xpsb_surface_desc *s, int mode) +{ + /* The first three are Xpsb.so's own Render enum, whose third mode is a + * clamp variant whose code 7 gl-re/textures.md section 1.2 records as + * never determined. Mirrored repeat is not in that enum at all - it is + * code 1, from psb_dri.so's translate_wrap at 0x2d280 - so it gets a + * mode of its own rather than borrowing the one nobody decoded. */ + /* The hardware's address-mode codes, indexed by the caller's mode. + * Named by the DDK as EURASIA_PDS_DOUTT0_ADDRMODE_*: 0 REPEAT, + * 1 FLIP (mirrored repeat), 2 CLAMP (clamp to edge), 3 FLIPCLAMP, + * 7 OGLCLAMP - legacy GL_CLAMP, which is what Xpsb.so's clampGL + * selects and what this table's third entry has always been. That + * entry stays for the callers that use it, but the Gallium driver no + * longer selects it: code 7 renders nothing on this part, and no + * vendor code chooses it either. See sgx_wrap_of(). + * Codes 4 to 6 are the border forms - CLAMPBDR 5, CLAMPBDRMEM 6, + * REPEATBDRMEM 4 - which nothing in the vendor stack selects; the + * two MEM forms need the border map in front of the texels, which + * the caller says it laid out with border_map. */ + static const uint32_t addr_code[8] = { 0, 2, 7, 1, 3, 5, 6, 4 }; + /* The filter field is two bits, not one: 0 POINT, 1 LINEAR, 2 ANISO, + * 3 ANISOPOINT (EURASIA_PDS_DOUTT0_MINFILTER_*). The ratio in + * ANISOCTL is only acted on when the filter itself names ANISO, so + * the two have to be set together. */ + static const uint32_t filter_code[4] = { 0, 1, 2, 3 }; + uint32_t base; + + if (mode != XPSB_TEXCTL_CAPTURE && mode != XPSB_TEXCTL_XPSB) { + errno = EINVAL; + return -1; + } + if (xpsb_surface_check(s)) return -1; + if (s->umode > 7 || s->vmode > 7 || s->smode > 7 || + s->minfilter > 3 || s->magfilter > 3) { + errno = EINVAL; + return -1; + } + /* A border-map mode without a map would read whatever precedes the + * texels as its border. The map's table is square and log2-sized, so + * that is the only shape it is offered on. */ + if (s->umode >= XPSB_WRAP_CLAMP_BORDER_MAP || + s->vmode >= XPSB_WRAP_CLAMP_BORDER_MAP || + s->smode >= XPSB_WRAP_CLAMP_BORDER_MAP) { + if (!s->border_map || !s->twiddled || s->w != s->h || + !xpsb_border_map_face_offset(s->w)) { + errno = EINVAL; + return -1; + } + } + + base = mode == XPSB_TEXCTL_XPSB ? 0x001e0002u : 0x03fe0000u; + *w0 = base + | ((addr_code[s->umode] << 6) & 0x000001c0u) + | ((addr_code[s->vmode] << 3) & 0x00000038u) + | ((filter_code[s->minfilter] << 10) & 0x00000c00u) + | ((filter_code[s->magfilter] << 12) & 0x00003000u) + | (xpsb_format_bpp(s->format) == 1 ? 0x40000000u : 0u) + | (s->chanrep ? 0x40000000u : 0u) + | (s->gamma ? 0x08000000u : 0u); + /* The slice axis, [2:0]. Left at the captured word's zero for a 2D + * texture, which is REPEAT and which no 2D sample consults. */ + if (s->depth > 1) + *w0 = (*w0 & ~0x00000007u) | (addr_code[s->smode] & 7u); + /* Bits 20:17 are the top level index, and the 0xF both bases carry is + * that field saturated - which is how the blob spells "not mipmapped". + * A real chain replaces it with its own last level. */ + if (s->nlevels > 1) { + *w0 = (*w0 & ~0x001e0000u) + | (((s->nlevels - 1u) & 0xfu) << 17); + if (s->mipfilter) + *w0 |= 0x00000200u; + } + /* The LOD adjust, DOUTT0[26:21]. Left alone at the field's zero so the + * captured word is reproduced exactly when nothing asks for a bias. */ + if (s->lod_bias && s->lod_bias != XPSB_LOD_BIAS_ZERO) + *w0 = (*w0 & ~0x07e00000u) | ((s->lod_bias & 0x3fu) << 21); + /* Anisotropy is two fields, not one: the ratio at DOUTT0[16:14] and an + * anisotropic mode in each filter field. The DDK writes both together + * and the ratio on its own selects nothing, so this does not offer a + * way to set one without the other. Filter code 2 is ANISO - linear + * within the footprint - against 3, ANISOPOINT. */ + if (s->aniso) { + *w0 = (*w0 & ~0x0001c000u) | ((s->aniso & 7u) << 14); + if (s->minfilter) + *w0 = (*w0 & ~0x00000c00u) | (2u << 10); + if (s->magfilter) + *w0 = (*w0 & ~0x00003000u) | (2u << 12); + } + /* And the highest level the unit may sample, which shares the field + * the chain's last level went into above: a sampler asking for fewer + * levels than the texture carries clamps here rather than by + * describing a shorter chain, so the levels stay addressable. */ + if (s->nlevels > 1 && s->max_level_set && + s->max_level < s->nlevels - 1u) + *w0 = (*w0 & ~0x001e0000u) + | ((s->max_level & 0xfu) << 17); + if (s->twiddled) { + /* Texture type zero, the twiddled 2D form, unless the caller + * names another - a cube is type 2 with the same log2 sizes, + * its faces found by the unit from one base address. A volume + * is type 1 and carries its depth in the same form at + * [15:12]. */ + uint32_t type = s->depth > 1 ? XPSB_TEXTYPE_3D : s->textype; + + *w1 = ((type & 7u) << 29) + | ((uint32_t)s->format << 24) + | ((uint32_t)exact_log2(s->w) << 16) + | ((uint32_t)exact_log2(s->h)); + if (s->depth > 1) + *w1 |= (uint32_t)exact_log2(s->depth) << 12; + } else + *w1 = XPSB_SURF_FMT(s->format) + | (((s->w - 1u) << 12) & 0x00fff000u) + | ((s->h - 1u) & 0x00000fffu); + return 0; +} + +/* A GL LOD bias as the six-bit field the TAG reads. Clamped rather than + * refused: GL lets a caller ask for any bias and expects the implementation's + * own range to apply, which is what MAX_TEXTURE_LOD_BIAS reports. */ +uint32_t xpsb_lod_bias_code(float bias) +{ + float steps; + + if (!(bias > XPSB_LOD_BIAS_MIN)) /* also catches NaN */ + return 0u; + if (bias > XPSB_LOD_BIAS_MAX) + bias = XPSB_LOD_BIAS_MAX; + steps = (bias - XPSB_LOD_BIAS_MIN) * 8.0f + 0.5f; + return (uint32_t)steps & 0x3fu; +} + +int xpsb_tex_state(uint32_t *w0, uint32_t *w1, + const struct xpsb_surface_desc *s) +{ + return xpsb_tex_state_mode(w0, w1, s, XPSB_TEXCTL_CAPTURE); +} + +int xpsb_heap_set_dest(uint32_t *h, const struct xpsb_surface_desc *d) +{ + unsigned bpp = xpsb_format_bpp(d->format); + + if (xpsb_surface_check(d)) return -1; + /* The line-stride register counts dwords, not pixels; the two are the + * same only on a four-byte target, which is every target the capture + * had. */ + (void)bpp; + heap_dest_fields(h, (int)d->w, (int)d->h, d->stride / 4u, d->format); + return 0; +} + +/* Buffer 3 (MEM_RASTGEOM) holds three geometry records the rasteriser uses for + * the clear and the present blit. Screen coordinates are 12.4 fixed point + * biased 0x4000, so 0x48004000 is (128.0, 0.0). Each record is a header, a + * vertex count, then three (packed_xy, 1.0f) pairs. Only the coordinates + * depend on render size; the header words are heap offsets. */ +#define FIX124(v) ((uint32_t)(((int)((v) * 16.0f)) + 0x4000)) +#define XY(x, y) ((FIX124(x) << 16) | FIX124(y)) + +static uint32_t f32_bits(float v) +{ + uint32_t u; + + memcpy(&u, &v, sizeof u); + return u; +} + +/* The record's own triangle is the capture's: (0,0), (w,0), (0,h) with (u,v) + * of (0,0), (1,0), (0,1). Growing it to contain the whole render was tried and + * measured on the part - a triangle covering the surface four times over + * changes nothing at all, and collapsing it to a point changes everything - so + * the object's extent decides nothing here and the capture stands. What moves + * it is SGX_BG_SCALE below, kept as an instrument rather than a shape. */ + +/* Records 0 and 1 clear the target; record 2 is the present blit. The two + * header words are heap addresses, so off0/offc are both the placeholder the + * record carries and the pre_add of the relocation that overwrites it. */ +static const struct { unsigned at, vtx; uint32_t h1, off0, offc, ox, oy; } +rast_rec[3] = { + { 0x000, 0x00b, 0x00030000, 0x00100, 0x000c0, 0, 0 }, + { 0x014, 0x01f, 0x00030001, 0x00160, 0x00120, 0, 0 }, + { 0x400, 0x40b, 0x00030001, 0x40100, 0x400c0, + XPSB_PRESENT_X, XPSB_PRESENT_Y }, +}; + +#define RAST_H0(o) (((XPSB_HEAP_ADDR + (o)) >> 4) & 0x00ffffffu) + +/* SGX_BG_SCALE and SGX_BG_OFF move the two background objects' triangle + * without moving anything else, so a sweep can say whether the region a + * load-back reaches follows this record at all. SCALE is the leg as a multiple + * of the surface - 1.0 is the capture, 2.0 the triangle that contains the + * whole render - and OFF="dx,dy" translates it in pixels. The coordinates + * follow the positions in both cases, so a picture that moves is the record + * and a picture that does not is something else. + * + * Diagnostic only: unset, the generator behaves exactly as it does below. */ +static void rast_user_tri(float fw, float fh, float *sx, float *sy, + float *dx, float *dy) +{ + const char *es = XPSB_ENVS("SGX_BG_SCALE"), *eo = XPSB_ENVS("SGX_BG_OFF"); + + *dx = *dy = 0.0f; + if (eo) + sscanf(eo, "%f,%f", dx, dy); + if (!es) + return; + *sx = (float)strtod(es, NULL) * fw; + *sy = (float)strtod(es, NULL) * fh; +} + +void xpsb_gen_rastgeom_from(uint32_t *b, int w, int h, unsigned first) +{ + unsigned r; + + for (r = first; r < 3; r++) { + uint32_t *p = b + rast_rec[r].at; + float ox = (float)rast_rec[r].ox, oy = (float)rast_rec[r].oy; + float fw = (float)w, fh = (float)h; + float a = fw, c = fh, dx, dy; + + rast_user_tri(fw, fh, &a, &c, &dx, &dy); + ox += dx; + oy += dy; + p[0] = RAST_H0(rast_rec[r].off0); + p[1] = rast_rec[r].h1; + p[2] = 0x0c000000u | RAST_H0(rast_rec[r].offc); + /* The vertex format word: TSP size 2, one (u,v) pair a vertex + * (2 << EURASIA_PARAM_VF_TSP_SIZE_SHIFT, sgxdefs.h:8075). */ + b[rast_rec[r].vtx - 2] = 0x00000002; + p = b + rast_rec[r].vtx; + /* Positions and (u,v) at +3..+8 are one affine map - moving + * one without the other leaves every pixel sampling a clamped + * corner - so the legs and the coordinates scale together. */ + p[0] = XY(ox, oy); p[1] = 0x3f800000; + p[2] = XY(ox + a, oy); p[3] = 0x3f800000; + p[4] = XY(ox, oy + c); p[5] = 0x3f800000; + /* Record 0 samples nothing, so its coordinate slot stays as + * the capture left it. All six of the others are written, not + * only the two the capture has non-zero, because a translated + * triangle has no zero coordinate left. */ + if (r >= 1) { + uint32_t *t = b + rast_rec[r].at + 3; + + t[0] = f32_bits(dx / fw); t[1] = f32_bits(dy / fh); + t[2] = f32_bits((dx + a) / fw); t[3] = f32_bits(dy / fh); + t[4] = f32_bits(dx / fw); t[5] = f32_bits((dy + c) / fh); + } + } +} + +/* The three records as the hardware will read them, decoded. Called at submit + * with the mapped buffer, so what it prints is what reached memory and not + * what the generator meant to write - which is the difference between "the + * region does not follow the record" and "the record never arrived". */ +void xpsb_rastgeom_dump(FILE *f, const uint32_t *b) +{ + static const char *name[3] = { "clear", "loadback", "present" }; + unsigned r, i; + + if (!f || !b) + return; + for (r = 0; r < 3; r++) { + const uint32_t *p = b + rast_rec[r].at; + const uint32_t *v = b + rast_rec[r].vtx; + + fprintf(f, "sgx: rastgeom %u (%s) hdr %08x %08x %08x vf %08x\n", + r, name[r], p[0], p[1], p[2], b[rast_rec[r].vtx - 2]); + for (i = 0; i < 3; i++) { + float u, uv; + uint32_t xy = v[i * 2]; + + memcpy(&u, p + 3 + i * 2, sizeof u); + memcpy(&uv, p + 4 + i * 2, sizeof uv); + fprintf(f, "sgx: v%u xy %08x = (%.2f,%.2f) z %08x " + "uv (%.4f,%.4f)\n", i, xy, + ((double)(int)((xy >> 16) & 0xffffu) - 0x4000) / 16.0, + ((double)(int)(xy & 0xffffu) - 0x4000) / 16.0, + v[i * 2 + 1], (double)u, (double)uv); + } + } +} + +void xpsb_rastgeom_clear_none(uint32_t *b) +{ + unsigned r; + + for (r = 0; r < 2; r++) { + uint32_t *p = b + rast_rec[r].vtx; + uint32_t at = XY(0.0f, 0.0f); + + p[0] = at; p[2] = at; p[4] = at; + } +} + +void xpsb_gen_rastgeom(uint32_t *b, int w, int h) +{ + xpsb_gen_rastgeom_from(b, w, h, 0); +} + +/* Buffer 5 opens with the draw-record array: four records describing the + * draws in the frame. The first three are 7 dwords (stride 0x1C); the last is + * a 4-dword terminator whose word 2 is 0xc0000000. + * + * +0x00 0x?2001xxx record header, low 12 bits a heap offset + * +0x04 0x0c02a2NN NN counts down 3,2,4,1 across the records + * +0x08 draw command: 0x81400000 | index_count + * +0x0c index-buffer address (relocated) + * +0x14 primitive-descriptor address + * +0x18 0x02000103 / 0x06000303 draw kind + * + * Record 2 is the user geometry: the caller overwrites its draw command with + * the live index count and the kernel relocates its index pointer, which is + * why XPSB_DRAW_CMD_OFF is 0x40 - word 2 of the third record. + * + * Words 0, 3 and 5 are the three relocated dwords, so their heap offsets live + * in the descriptor and feed both the placeholder here and the pre_add of the + * matching relocation. Descriptor 3 is the terminator. */ +const struct xpsb_draw_desc xpsb_draw_descs[4] = { + { 0x40000000, 0x0300, 3, 0x01fc, 0x0220, 0x02000103, 6 }, + { 0x40000000, 0x44e0, 2, 0x4464, 0x4480, 0x02000103, 6 }, + { 0x40000000, 0x4560, 4, 0x03e4, 0x03a0, 0x06000303, 6 }, + { 0x60000000, 0x0180, 1, 0, 0, 0, 0 }, +}; + +static void put_draw_record(uint32_t *p, const struct xpsb_draw_desc *d, + int terminator) +{ + p[0] = d->hdr_bg | (((XPSB_HEAP_ADDR + d->state) >> 4) & 0x0fffffffu); + p[1] = 0x0c02a200u | d->seq; + if (terminator) { + p[2] = 0xc0000000u; + p[3] = 0; + return; + } + p[2] = XPSB_DRAW_CMD_TAG | d->nidx; + p[3] = XPSB_HEAP_ADDR + d->idx; + p[4] = 0; + p[5] = ((XPSB_HEAP_ADDR + d->prim) >> 4) & 0x0fffffffu; + p[6] = d->kind; +} + +unsigned xpsb_gen_draw_records_from(uint32_t *b, unsigned first) +{ + unsigned r, n = 0; + + if (first > XPSB_USER_DRAW) return 0; + for (r = first; r < XPSB_NUM_DRAW; r++) + put_draw_record(b + n++ * XPSB_DRAW_STRIDE, + &xpsb_draw_descs[r], 0); + put_draw_record(b + n * XPSB_DRAW_STRIDE, &xpsb_draw_descs[3], 1); + return n + 1; +} + +unsigned xpsb_gen_draw_records_n(uint32_t *b, unsigned first, unsigned ntex) +{ + unsigned n = xpsb_gen_draw_records_from(b, first); + + if (!n || ntex < 1 || ntex > XPSB_NTEX_MAX) return 0; + b[(XPSB_USER_DRAW - first) * XPSB_DRAW_STRIDE + 6] = + XPSB_DRAW_KIND((xpsb_vtx_stride(ntex) + 3) / 4); + return n; +} + +void xpsb_gen_draw_records(uint32_t *b) +{ + xpsb_gen_draw_records_from(b, 0); +} + +unsigned xpsb_draw_cmd_off(unsigned first) +{ + return (XPSB_USER_DRAW - first) * XPSB_DRAW_STRIDE * 4 + 8; +} + +/* The vertex record is 11 floats: position, colour, texture coordinate. z, w + * and q are the constants drmquad.c feeds for a screen-space quad. Corners run + * (x0,y0) (x1,y0) (x1,y1) (x0,y1), which is drmquad.c:749-750's winding, and + * the indices triangulate it the way the driver does. */ +unsigned xpsb_vtx_stride(unsigned ntex) +{ + return 8 + 3 * ntex; +} + +unsigned xpsb_gen_quad_n(float *vtx, uint16_t *idx, const struct xpsb_quad *q, + unsigned ntex, unsigned *nvtx) +{ + static const uint16_t tri[6] = { 0, 1, 3, 1, 2, 3 }; + const float x[4] = { q->x0, q->x1, q->x1, q->x0 }; + const float y[4] = { q->y0, q->y0, q->y1, q->y1 }; + const float u[4] = { q->u0, q->u1, q->u1, q->u0 }; + const float v[4] = { q->v0, q->v0, q->v1, q->v1 }; + const float m[4] = { q->m0, q->m1, q->m1, q->m0 }; + const float n[4] = { q->n0, q->n0, q->n1, q->n1 }; + unsigned stride = xpsb_vtx_stride(ntex), k; + + if (ntex < 1 || ntex > XPSB_NTEX_MAX) return 0; + for (k = 0; k < 4; k++) { + float *o = vtx + (size_t)k * stride; + o[0] = x[k]; o[1] = y[k]; o[2] = 0.5f; o[3] = 1.0f; + o[4] = q->r; o[5] = q->g; o[6] = q->b; o[7] = q->a; + o[8] = u[k]; o[9] = v[k]; o[10] = 1.0f; + if (ntex > 1) { + o[11] = m[k]; o[12] = n[k]; o[13] = 1.0f; + } + } + if (idx) memcpy(idx, tri, sizeof tri); + if (nvtx) *nvtx = 4; + return 6; +} + +unsigned xpsb_gen_quad(float *vtx, uint16_t *idx, const struct xpsb_quad *q, + unsigned *nvtx) +{ + return xpsb_gen_quad_n(vtx, idx, q, 1, nvtx); +} + +/* Sixteen USSE programs in 32-byte slots, carried from test/drmcube.c as the + * 64-bit instruction words they are there. Only two are parameterised: the + * clear colour's 21-bit immediate, and the fragment slot, which the composite + * path overwrites. */ +unsigned xpsb_gen_usse(uint32_t *u, uint32_t clear) +{ + static const struct { unsigned off; uint64_t insn[4]; } + slots[XPSB_USSE_NSLOT] = { + { 0x0000, { 0xfb20000484208180ull, 0xfb24004481200080ull } }, + { 0x0020, { 0xf834800000000000ull } }, + { 0x0040, { 0xfca7f18100000000ull } }, /* clear colour, below */ + { 0x0060, { 0x28851001a0000000ull } }, + { 0x0080, { 0x28a11001a0000000ull, 0xfb274000a0200100ull } }, + { 0x00a0, { 0x28a13001a0000000ull, 0xfb275000a0200000ull } }, + { 0x00c0, { 0xfca7f1810000ff00ull } }, + { 0x00e0, { 0x28a1a001a0000000ull, 0xfb274000a0200580ull } }, + { 0x0100, { 0x81840005a0000080ull } }, /* fragment */ + { 0x0140, { 0x28a1a001a0000000ull, 0xfb275000a0200000ull } }, + { 0x0160, { 0x28a13001a0000000ull, 0xfb275000a0200000ull } }, + { 0x0180, { 0x28a15001a0000000ull, 0xfb274000a0200300ull } }, + { 0x01a0, { 0x28a1e001a0000000ull, 0xfb274000a0200780ull } }, + { 0x2000, { 0xfb20000484208180ull, 0xfb24004481200080ull } }, + { 0x2020, { 0xf834800000000000ull } }, + { 0x2040, { 0x28851001a0000000ull } }, + }; + unsigned k, j; + + memset(u, 0, XPSB_USSE_DW * 4); + for (k = 1; k <= XPSB_USSE_STATE_MAX; k++) + xpsb_usse_state_copy(u + xpsb_usse_state_copy_off(k) / 4, k); + for (k = 0; k < XPSB_USSE_NSLOT; k++) { + uint32_t *w = u + slots[k].off / 4; + uint64_t i0 = slots[k].insn[0]; + + /* The clear colour is a full 32-bit immediate inside this + * instruction, and it is spread across three fields: bits + * 20:0 in the low word, 25:21 at w1[8:4] and 31:26 at + * w1[17:12]. Only the first was ever written, so the top three + * bits of the channel that lands at <<16 were dropped - a + * clear of 0xff there stored as 0x1f. The captured default + * 0x1a1a26 hid it, its top byte being under 0x20, and the + * instruction's own high fields already carry the 0xff alpha, + * which is what identified them. */ + if (slots[k].off == 0x0040) { + uint32_t imm = 0xff000000u | (clear & 0x00ffffffu); + uint32_t w1 = (uint32_t)(i0 >> 32) & ~0x0003f1f0u; + + w1 |= ((imm >> 21) & 0x1fu) << 4; + w1 |= ((imm >> 26) & 0x3fu) << 12; + i0 = ((uint64_t)w1 << 32) | (imm & 0x001fffffu); + } + w[0] = (uint32_t)i0; + w[1] = (uint32_t)(i0 >> 32); + for (j = 1; j < 4; j++) { + w[j * 2] = (uint32_t)slots[k].insn[j]; + w[j * 2 + 1] = (uint32_t)(slots[k].insn[j] >> 32); + } + } + return XPSB_USSE_DW; +} + +unsigned xpsb_usse_video(uint32_t *u, uint32_t fourcc) +{ + const uint32_t *prog; + unsigned n = 0; + + prog = xpsb_vidshader(fourcc, &n); + if (!prog || n > XPSB_USSE_VID_SLOT) + return 0; + memset(u + XPSB_USSE_VID_OFF / 4, 0, XPSB_USSE_VID_SLOT * 4); + memcpy(u + XPSB_USSE_VID_OFF / 4, prog, n * 4); + return n; +} + +/* Instructions 0, 1 and 2 of the packed program are the only PCKUNPCKs in it, + * at dwords 0, 2 and 4; their high words are untouched. Checked against + * tools/isa-usse/usse-dis, which reads 0xa0400082 as pa1 and reports no other + * field changed. */ +int xpsb_usse_video_pa(uint32_t *u, uint32_t fourcc, unsigned pa) +{ + uint32_t *w = u + XPSB_USSE_VID_OFF / 4; + unsigned i; + + if (xpsb_vidshader_offset(fourcc) != XPSB_USSE_OFF_PACKED || + pa > (XPSB_PCK_SRC1_MASK >> XPSB_PCK_SRC1_SHIFT)) { + errno = EINVAL; + return -1; + } + for (i = 0; i < 3; i++) + w[2 * i] = (w[2 * i] & ~XPSB_PCK_SRC1_MASK) | + (pa << XPSB_PCK_SRC1_SHIFT); + return 0; +} + +void xpsb_heap_set_video(uint32_t *h, const float *conv) +{ + h[XPSB_HEAP_USE_TEMPS] = conv ? XPSB_VID_TEMPS : 0; + if (conv) + memcpy(h + XPSB_VID_CONST_DW, conv, + XPSB_VID_CONST_N * sizeof *conv); + else + memset(h + XPSB_VID_CONST_DW, 0, + XPSB_VID_CONST_N * sizeof *conv); +} + +/* offset / 16 in bits [18:8] is read off the captured word, 0x00181025 for the + * program at 0x100: derived, but the relocation's 0x0007ff00 mask replaces the + * whole field before the stream is read. */ +void xpsb_heap_set_frag_use(uint32_t *h, uint32_t off) +{ + uint32_t o = off ? off : XPSB_USSE_FRAG_OFF; + + h[XPSB_PRI_PDS_DW] = (h[XPSB_PRI_PDS_DW] & ~0x0007ff00u) | + (((o >> 4) << 8) & 0x0007ff00u); +} + +/* Both constant routes into the secondary attribute bank are the same PDS + * program shape, so the video path's emitter builds the composite scalar too: + * one constant is its n == 1 case, eleven its DMA case. */ +unsigned xpsb_gen_sec_pds(uint32_t *out, const float *consts, unsigned n, + uint32_t const_addr, unsigned *dsize) +{ + struct xpsb_vidshader_sa_pds pds; + + if (!n || xpsb_vidshader_emit_sa_pds(&pds, consts, n, const_addr)) + return 0; + memcpy(out, pds.prog, pds.n_prog * 4); + if (dsize) + *dsize = pds.data_dwords; + return pds.n_prog; +} + +void xpsb_heap_set_sec_pds(uint32_t *h, uint32_t off, unsigned dsize, + unsigned nattr) +{ + uint32_t at = off ? off : XPSB_SEC_PDS_NULL; + /* Blocks of 128 bytes, and nattr counts registers. */ + uint32_t blocks = (nattr * 4u + 0x7fu) >> 7; + + uint32_t *ctl = h + xpsb_heap_pds_dw(h, XPSB_PDS_W_CTL); + + h[xpsb_heap_pds_dw(h, XPSB_PDS_W_SEC)] = + (((XPSB_HEAP_ADDR + at) >> 4) & 0x00ffffffu) | + ((uint32_t)dsize << 24); + *ctl = (*ctl & ~XPSB_SA_BLOCK_MASK) | ((blocks << 18) & XPSB_SA_BLOCK_MASK); +} + +void xpsb_usse_default_frag(uint32_t *slot) +{ + memset(slot, 0, XPSB_USSE_SLOT_DW * 4); + slot[0] = XPSB_FRAG_DEFAULT_LO; + slot[1] = XPSB_FRAG_DEFAULT_HI; +} + +/* The composite shader is two SOP2 instructions, low word first, in the same + * 32-byte slot the captured single-instruction program occupies. */ +int xpsb_usse_composite(uint32_t *slot, int op, + const struct xpsb_shader_flags *fl) +{ + uint32_t w[4]; + + if (!fl || xpsb_composite_shader(op, fl, w)) { + errno = EINVAL; + return -1; + } + memset(slot, 0, XPSB_USSE_SLOT_DW * 4); + memcpy(slot, w, sizeof w); + return 0; +} + +/* ---- relocation emitters ---- + * + * The 77 TA records are not a repeating pattern: most of them address heap + * state blocks whose contents are captured constants, and those are carried + * verbatim in the order they appear. Only three classes vary, and each is + * derived from the same descriptor as the dword it patches: + * + * RC_DRAW three records per draw block, at dwords 7r+0, +3 and +5 + * RC_CLEAR two records per clearing rastgeom record + * RC_TEX one record per texture unit, at heap dword 0xda + 2u + * RC_PRI the primary PDS binding, whose background is that program's + * data-segment size and so follows the texture unit count + * RC_FRAG the user draw's fragment USE task, whose pre_add is the byte + * offset of the program in buffer 2 and whose arg0 is its size + * + * arg0 is the size psb_grab_use_base() reserves in the USE window. All thirteen + * captured USE relocations carry eight bytes per instruction of the program + * they name - 0x8 for the one-instruction slots, 0x10 for the two-instruction + * ones - so it is the program size in bytes. + * + * RC_DEST and RC_TEX also carry the caller's surface offset as pre_add, which + * is the only reason a caller-supplied surface touches this table at all. */ +#define RC_BASE 0 +#define RC_SKIP 1 /* generated together with the preceding RC_DRAW */ +#define RC_DEST 2 +#define RC_CLEAR 3 +#define RC_DRAW 4 +#define RC_TERM 5 +#define RC_TEX 6 +#define RC_SEC 7 /* the pixel binding's secondary PDS program */ +#define RC_PRI 8 /* the pixel binding's primary PDS program */ +#define RC_FRAG 9 /* the user draw's fragment USE task, three records */ +/* The ISP background object, raster register 0x4c4: which rastgeom record + * paints a tile the tiler did not bin. Record 0 clears it; record 1 loads it + * back from the colour buffer, which is what a frame drawing over an existing + * picture needs. */ +#define RC_BG 10 +/* The state block's descriptor names the block by address, and the block + * moves once it outgrows its home - see xpsb_heap_state_base(). */ +#define RC_STATE 11 +/* Its DOUTU, three records: the copy program follows the block's size. */ +#define RC_STATE_USE 12 + +static const struct { unsigned at, cls, arg; } ta_reloc_class[] = { + { 0, RC_DEST, 0 }, + { 11, RC_CLEAR, 0 }, { 12, RC_CLEAR, 0 }, + { 16, RC_DEST, 0 }, + { 17, RC_CLEAR, 1 }, { 18, RC_CLEAR, 1 }, + { 29, RC_BG, 0 }, + { 45, RC_DRAW, 0 }, { 46, RC_SKIP, 0 }, { 47, RC_SKIP, 0 }, + { 48, RC_TEX, 0 }, + { 49, RC_FRAG, 0 }, { 50, RC_FRAG, 0 }, { 51, RC_FRAG, 0 }, + { 64, RC_DRAW, 1 }, { 65, RC_SKIP, 0 }, { 66, RC_SKIP, 0 }, + { 67, RC_SEC, 0 }, { 68, RC_PRI, 0 }, + { 69, RC_STATE, 0 }, + { 70, RC_STATE_USE, 0 }, { 71, RC_STATE_USE, 0 }, { 72, RC_STATE_USE, 0 }, + { 73, RC_DRAW, 2 }, { 74, RC_SKIP, 0 }, { 75, RC_SKIP, 0 }, + { 76, RC_TERM, 0 }, +}; + +static unsigned reloc_class(unsigned at, unsigned *arg) +{ + unsigned i; + + for (i = 0; i < sizeof ta_reloc_class / sizeof ta_reloc_class[0]; i++) + if (ta_reloc_class[i].at == at) { + *arg = ta_reloc_class[i].arg; + return ta_reloc_class[i].cls; + } + return RC_BASE; +} + +static void put_draw_relocs(struct xpsb_reloc *o, unsigned slot, + const struct xpsb_draw_desc *d, int terminator) +{ + unsigned w = slot * XPSB_DRAW_STRIDE; + + memset(o, 0, (terminator ? 1 : 3) * sizeof o[0]); + o[0].reloc_op = XPSB_RELOC_OP_OFFSET; + o[0].where = w; + o[0].buffer = XPSB_BUF_HEAP; + o[0].mask = 0x0fffffff; + o[0].shift = 0x40000; + o[0].pre_add = d->state; + o[0].background = d->hdr_bg; + o[0].dst_buffer = XPSB_BUF_VTX; + if (terminator) return; + + o[1].reloc_op = XPSB_RELOC_OP_OFFSET; + o[1].where = w + 3; + o[1].buffer = XPSB_BUF_HEAP; + o[1].mask = 0xffffffff; + o[1].pre_add = d->idx; + o[1].dst_buffer = XPSB_BUF_VTX; + + o[2].reloc_op = XPSB_RELOC_OP_OFFSET; + o[2].where = w + 5; + o[2].buffer = XPSB_BUF_HEAP; + o[2].mask = 0x0fffffff; + o[2].shift = 0x40000; + o[2].pre_add = d->prim; + o[2].dst_buffer = XPSB_BUF_VTX; +} + +unsigned xpsb_gen_ta_relocs(struct xpsb_reloc *out, + const struct xpsb_reloc_cfg *c) +{ + unsigned i, n = 0, cls, arg; + + /* Two clears or none: dropping only one leaves a rastgeom record with + * no draw. Past two texture units the descriptor array spills past the + * address list and the attribute DMA splits, which adds a relocation + * this emitter does not produce. */ + if (c->first_draw != 0 && c->first_draw != XPSB_USER_DRAW) return 0; + if (c->ntex < 1 || c->ntex > XPSB_NTEX_MAX) return 0; + + for (i = 0; i < XPSB_NUM_TA_RELOCS; i++) { + cls = reloc_class(i, &arg); + switch (cls) { + case RC_SKIP: + break; + case RC_CLEAR: + if (c->first_draw) break; + out[n++] = xpsb_ta_relocs[i]; + break; + case RC_DEST: + out[n] = xpsb_ta_relocs[i]; + out[n++].pre_add = c->dest_offset; + break; + case RC_DRAW: + if (arg < c->first_draw) break; + put_draw_relocs(out + n, arg - c->first_draw, + &xpsb_draw_descs[arg], 0); + n += 3; + break; + case RC_TERM: + put_draw_relocs(out + n, XPSB_NUM_DRAW - c->first_draw, + &xpsb_draw_descs[3], 1); + n += 1; + break; + case RC_SEC: + out[n] = xpsb_ta_relocs[i]; + if (c->pds_dw) + out[n].where = c->pds_dw + XPSB_PDS_W_SEC; + if (c->sec_pds_off) { + out[n].pre_add = c->sec_pds_off; + out[n].background = + (uint32_t)c->sec_pds_dwords << 24; + } + n++; + /* The DOUTD source is data dword 0 of that program, and + * it is a heap address: the same whole-dword form as the + * captured DMA descriptors at heap+0x60, +0x88 and + * +0xc0, each of which carries one of these. */ + if (c->sec_const_off) { + memset(&out[n], 0, sizeof out[n]); + out[n].reloc_op = XPSB_RELOC_OP_OFFSET; + out[n].where = c->sec_pds_off / 4; + out[n].buffer = XPSB_BUF_HEAP; + out[n].mask = 0xffffffff; + out[n].pre_add = c->sec_const_off; + out[n].dst_buffer = XPSB_BUF_HEAP; + n++; + } + break; + case RC_FRAG: + out[n] = xpsb_ta_relocs[i]; + /* The DOUTU words are the primary program's own data + * dword 0, so they follow it when it moves. */ + if (c->pri_pds_off) + out[n].where += c->pri_pds_off / 4 - + XPSB_PRI_PDS_DW; + if (c->frag_use_off) { + out[n].pre_add = c->frag_use_off; + out[n].arg0 = c->frag_use_size; + } + n++; + break; + case RC_BG: + out[n] = xpsb_ta_relocs[i]; + if (c->bg_load) + out[n].pre_add = 0x50; + n++; + break; + case RC_STATE: + out[n] = xpsb_ta_relocs[i]; + if (c->state_off) + out[n].pre_add = c->state_off; + n++; + break; + case RC_STATE_USE: + out[n] = xpsb_ta_relocs[i]; + if (c->state_use_off) { + out[n].pre_add = c->state_use_off; + out[n].arg0 = c->state_use_size; + } + n++; + break; + case RC_PRI: + out[n] = xpsb_ta_relocs[i]; + if (c->pds_dw) + out[n].where = c->pds_dw + XPSB_PDS_W_PRI; + if (c->pri_pds_off) + out[n].pre_add = c->pri_pds_off; + /* The issue list's own count when it built the + * program: sizing three issues through the plain + * generator returns zero, which declares an empty + * program over a real one. */ + out[n++].background = + (c->pri_pds_dwords ? c->pri_pds_dwords : + pri_pds_dwords(c->ntex)) << 24; + break; + case RC_TEX: { + /* One per address slot the primary program carries, + * not one per bound unit: a slot no unit patches keeps + * its placeholder, which is mapped nowhere, and the + * render faults on it with the cache as the requestor. + * Those slots take unit 0's buffer - it is always + * bound and nothing reads the texel. */ + unsigned u, ns = c->ntexslot > c->ntex ? + c->ntexslot : c->ntex; + + if (ns > XPSB_MAX_TEX) + ns = XPSB_MAX_TEX; + for (u = 0; u < ns; u++) { + unsigned b = u < c->ntex ? u : 0; + unsigned at = c->nissue ? c->tex_issue[u] : u; + + out[n] = xpsb_ta_relocs[i]; + out[n].where = xpsb_texaddr_dw_at( + c->pri_pds_off ? c->pri_pds_off : + XPSB_PRI_PDS_OFF, at); + out[n].buffer = XPSB_BUF_TEX + b; + out[n++].pre_add = u < c->ntex ? + c->tex_offset[u] : 0; + } + break; + } + default: + out[n++] = xpsb_ta_relocs[i]; + break; + } + } + return n; +} + +/* The raster pass has no draw records: it blits the render target, which is at + * validate position 3 there, so only that one record follows the caller's + * destination offset. */ +unsigned xpsb_gen_raster_relocs(struct xpsb_reloc *out, + const struct xpsb_reloc_cfg *c) +{ + unsigned i; + + for (i = 0; i < XPSB_NUM_RAS_RELOCS; i++) { + out[i] = xpsb_raster_relocs[i]; + if (out[i].reloc_op == XPSB_RELOC_OP_OFFSET && + out[i].where == 0x01003a && out[i].buffer == 3) + out[i].pre_add = c->dest_offset; + /* The render pass's background object is not here: this is + * the present blit, whose own background object is record 2. + * Retargeting it would blit at (0,0) through the wrong scene + * block. The render pass is patched through RC_BG above. */ + } + return XPSB_NUM_RAS_RELOCS; +} + +/* SGX register/value pair streams. psb_memcpy_check() copies size>>3 pairs, so + * the stream is (reg,value) dwords; the kernel validates reg < 0x1000 against a + * deny-list. None of the TA values are relocated - they are literals. The only + * relocated value in either stream is raster reg 0x4c4, patched by the kernel. + * Values a relocation patches are written as captured constants. */ +unsigned xpsb_gen_ta_stream(uint32_t *p) +{ + static const uint32_t s[] = { + 0x0204, 0x00000000, 0x0218, 0x00400000, + 0x023c, 0x00000082, 0x0240, 0x358637bd, + 0x0244, 0x358637bd, 0x0250, 0x00000088, + 0x0238, 0x40002000, /* relocated: src buffer 5 */ + }; + memcpy(p, s, sizeof s); + return sizeof s / 4; /* dwords */ +} + +unsigned xpsb_gen_ta_raster_stream(uint32_t *p, int w, int h) +{ + uint32_t s[] = { + 0x0400, 0x00ffff80, 0x0404, 0x00000000, + 0x040c, 0x00000000, 0x0410, 0x00000000, /* tile counts, below */ + 0x0414, 0x00000100, 0x0418, 0x00000001, + 0x041c, 0x0da24260, 0x0420, 0x00000000, + 0x0424, 0x0000fbf4, 0x042c, 0x00000000, + 0x0480, 0x00450000, 0x0cb0, 0x80000000, /* width, below */ + 0x0484, 0x004da000, 0x0488, 0x004da000, /* relocated: buffer 6 */ + 0x048c, 0x004da000, 0x0490, 0x004da000, + 0x0494, 0x000000ff, 0x04c4, 0x02008000, /* relocated: buffer 3 */ + 0x04bc, 0x00000200, 0x04b8, 0x3f800000, + 0x04c8, 0x00000088, 0x04dc, 0x00000000, + 0x0800, 0x00000000, 0x0a5c, 0x20010020, /* relocated: buffer 0 */ + 0x0a60, 0x00000004, 0x0a64, 0x00004fff, + }; + /* The last tile, not how many there are. 0x0410 is + * EUR_CR_ISP_RENDBOX2 (sgx535defs.h:1189-1194) and the vendor writes + * it as ceil(x1 / 16) - 1 - "round down for left and top, round up for + * bottom and right", SGXTQ_SetupTransferRenderBox + * (sgxtransfer_utils.c:3259-3273). The present chunk below already + * writes an inclusive last tile from a capture, and the captured + * ta-raster values are the same arithmetic: 100 gave 6, 200 gave 12, + * 33 gave 2. + * + * It was a count here, which renders one tile column and one tile row + * past the surface. On a whole-surface render the extent clip at heap + * dword 0x003 discards them and they cost only work; move the surface + * address to a damage rectangle's origin and they land in the picture. + * The heap's tile shadow at 0x05a is a count and is not this. */ + s[7] = ((uint32_t)XPSB_LAST_TILE(w) << 16) | + (uint32_t)XPSB_LAST_TILE(h); + /* 0x480 is the ISP's Z load/store control - see + * xpsb_ta_raster_set_z_load() for the decode. [17:16] is the load + * format and [19:18] the store format, both 1 (I24ZI8S) here; [15:8] + * is the extent in sixteen-pixel tiles, one less than the width in + * them; bits 4 and 6 enable the store and bit 22 comes from the + * vendor's scene terminate. (w - 3) << 4 is that extent plus the + * constant 0xd0 low byte, for any width that divides by sixteen. */ + s[21] = 0x00450000u | ((uint32_t)((w - 3) << 4) & 0xffffu); + /* SGX_RASTER_SET=reg:val[,reg:val] overrides a register of this stream + * in place, to sweep for the one that makes a pass load the depth it + * stored rather than start from the background at 0x4b8. */ + { + const char *e = XPSB_ENVS("SGX_RASTER_SET"); + + while (e && *e) { + unsigned long reg = strtoul(e, (char **)&e, 0); + unsigned long val; + unsigned i; + + if (*e != ':') + break; + val = strtoul(e + 1, (char **)&e, 0); + for (i = 0; i + 1 < sizeof s / 4; i += 2) + if (s[i] == (uint32_t)reg) + s[i + 1] = (uint32_t)val; + if (*e == ',') + e++; + } + } + memcpy(p, s, sizeof s); + return sizeof s / 4; +} + +/* Multisampling. The field map and the provenance of every value are in + * xpsb_frame.h; this is the arithmetic. + * + * One nibble an axis, in sixteenths of a pixel, four samples a word. */ +static uint32_t msaa_word(const unsigned char *xy) +{ + uint32_t v = 0; + unsigned i; + + for (i = 0; i < 4; i++) + v |= ((uint32_t)(xy[i * 2] & 0xfu) << (i * 8)) | + ((uint32_t)(xy[i * 2 + 1] & 0xfu) << (i * 8 + 4)); + return v; +} + +unsigned xpsb_msaa_axis(unsigned samples) +{ + switch (samples) { + case XPSB_MSAA_1X: return 1; + case XPSB_MSAA_4X: return 2; + default: return 0; + } +} + +uint32_t xpsb_msaa_positions(unsigned samples) +{ + /* The pixel centre, which is what OpenGL mode asks for at one sample; + * the other three nibbles are unread with a single sample and the + * vendor leaves them zero. */ + static const unsigned char one[8] = { 8, 8, 0, 0, 0, 0, 0, 0 }; + /* The rotated grid the vendor programs at 2x2. */ + static const unsigned char four[8] = { 6, 2, 14, 6, 2, 10, 10, 14 }; + + switch (samples) { + case XPSB_MSAA_1X: return msaa_word(one); + case XPSB_MSAA_4X: return msaa_word(four); + default: return 0; + } +} + +int xpsb_ta_set_msaa(uint32_t *p, unsigned n, unsigned samples) +{ + unsigned axis = xpsb_msaa_axis(samples); + uint32_t pos = xpsb_msaa_positions(samples); + unsigned i; + + if (!p || !axis) + return -1; + for (i = 0; i + 1 < n; i += 2) { + if (p[i] == 0x0204) + p[i + 1] = (p[i + 1] & 0x3fffffffu) | + (axis > 1 ? 0xc0000000u : 0u); + else if (p[i] == 0x0250) + p[i + 1] = pos; + } + return 0; +} + +int xpsb_ta_raster_set_msaa(uint32_t *p, unsigned n, int w, int h, + unsigned samples) +{ + unsigned axis = xpsb_msaa_axis(samples); + uint32_t pos = xpsb_msaa_positions(samples); + /* Absolute, from the geometry, not scaled from what is in the stream: + * a setter that multiplied in place could not be applied twice and + * could not go back to one sample. RENDBOX2 takes the last tile + * inclusive, so the sample-tile count carries the same less one. */ + uint32_t tx = (uint32_t)((w + 15) / 16) * axis - 1u; + uint32_t ty = (uint32_t)((h + 15) / 16) * axis - 1u; + /* w - 3 underflows below three pixels, which is the whole of why this + * refused small targets: the bound was set at a tile rather than at + * the arithmetic's real edge. A target narrower than a tile still + * occupies one, and glamor renders into pixmaps of every size - a + * glyph, a cursor, an 8x8 tile - so refusing them dropped whole + * frames, which is what wmaker's start does through glamor. */ + uint32_t ext = ((w > 3 ? (((uint32_t)(w - 3)) >> 4) : 0u) + 1u) * + axis - 1u; + unsigned i; + + if (!p || !axis || w < 1 || h < 1) + return -1; + /* XPSB_MSAA_RENDBOX=0 leaves the render box in pixel tiles, to answer + * on the part whether the ISP walks the box in sample tiles - which is + * inferred here, not read out of the DDK. */ + { + const char *e = XPSB_ENVS("XPSB_MSAA_RENDBOX"); + + if (e && *e && strtoul(e, NULL, 0) == 0) { + tx = (uint32_t)XPSB_LAST_TILE(w); + ty = (uint32_t)XPSB_LAST_TILE(h); + } + } + /* Eight bits a side, so 2x2 cannot reach the 2048 one sample can. */ + if (tx > 0xffu || ty > 0xffu || ext > 0xffu) + return -1; + for (i = 0; i + 1 < n; i += 2) { + switch (p[i]) { + case 0x04c8: + p[i + 1] = pos; + break; + case 0x042c: + p[i + 1] = axis > 1 ? 1u : 0u; + break; + case 0x0410: + /* The ISP's render box, in tiles. Inferred rather than + * read: the DDK's only writer of RENDBOX is the + * transfer queue, which is never multisampled + * (sgxtransfer_utils.c:3270-3273). What is read is + * that the region-header array holds one entry per + * sample tile - FillLastRgnLUT() scales the screen + * maxima by the sample count + * (sgxrender_targets.c:271-272) and the array is + * allocated at tiles-per-macrotile times that count + * (sgxrender_targets.c:2408-2409, 2433-2435). The box + * bounds the walk over those headers, so it is taken + * to be in the same tiles they are. */ + p[i + 1] = (p[i + 1] & ~0x00ff00ffu) | (tx << 16) | ty; + break; + case 0x0480: + /* EUR_CR_ISP_ZLSCTL [15:8] is the extent in + * sixteen-pixel tiles less one. It doubles in x only: + * the vendor's own EGL depth setup multiplies the + * extent by two and the buffer by four when the + * surface is multisampled (srv_sgx.c:410-414, + * 430-434). */ + p[i + 1] = (p[i + 1] & ~0x0000ff00u) | (ext << 8); + break; + default: + break; + } + } + return 0; +} + +/* The planar video route delivers its nine coefficients as hardware filter + * taps, appended to this list; the packed route uses the secondary attributes + * instead and appends nothing. */ +/* The pixel extent the ISP iterates, as (h-1)<<12 | (w-1). */ +/* Narrow the tiler and the ISP to a rectangle of tiles. The stream is built + * for the whole surface, which is what a frame that clears needs; a frame that + * only touches part of an existing picture - which is most of what a display + * server asks for - rasters every tile of it for nothing. Registers 0x040c and + * 0x0410 are the first and last tile, as xpsb_gen_raster_stream() writes them + * for the present chunk. */ +void xpsb_ta_raster_set_range(uint32_t *p, unsigned n, unsigned tx0, + unsigned ty0, unsigned tx1, unsigned ty1) +{ + unsigned i; + + if (!p || tx1 < tx0 || ty1 < ty0) + return; + for (i = 0; i + 1 < n; i += 2) { + if (p[i] == 0x040c) + p[i + 1] = (tx0 << 16) | ty0; + else if (p[i] == 0x0410) + p[i + 1] = (tx1 << 16) | ty1; + } +} + +/* Bit 8 of raster register 0x4bc goes with the background object: the captured + * present chunk, whose background object is the load-back record, carries + * 0x300 where the render chunk's clearing one carries 0x200, and the + * out-of-memory chunk needed 0x300 on hardware. Found by scanning for the + * register rather than by index, so an edit to the stream cannot move it + * silently. */ +void xpsb_ta_raster_set_bg_load(uint32_t *p, unsigned n, int on) +{ + unsigned i; + + if (!p) + return; + for (i = 0; i + 1 < n; i += 2) + if (p[i] == 0x04bc) { + p[i + 1] = (p[i + 1] & 0xffu) | (on ? 0x300u : 0x200u); + return; + } +} + +/* Register 0x0480 is the ISP's Z load/store control. Its layout, and what is + * still missing from it. + * + * psb_dri.so builds the word at 0x2ab70: ((depth_rb->width >> 4) - 1) << 8, + * then 0x50000 for a four-byte depth buffer, 0xa0000 for a narrower one and a + * literal 0xf0000 when there is no depth buffer at all; psb_scene_terminate + * (0x2a3c2) ORs 0x004000d0 into the emitted word. The captured 0x004507d0 at + * 128 pixels and this generator's (w - 3) << 4 are the same number for any + * width that divides by sixteen - the low byte is that 0xd0, not part of the + * extent - so the stream always carries the terminate bits. + * + * Measured against the depth buffer read back after a render, with + * SGX_DEPTH_DUMP, at 64x64 and a window depth of 0.25, then confirmed field + * by field against the DDK's EUR_CR_ISP_ZLSCTL for this core: + * + * bit 0 LOADTILED: the Z/S buffer in memory is tiled (ours is linear). + * bit 1 SLOADEN, master stencil load enable. + * bit 2 ZLOADEN, master depth load enable. + * bit 3 STORETILED. + * bit 4 ZSTOREEN, master depth store enable. + * bit 5 FORCEZLOAD: load every region, marked in its header or not. + * bit 6 FORCEZSTORE: store every region likewise. + * bit 7 SSTOREEN, master stencil store enable. + * [15:8] extent in sixteen-pixel tiles, one less than the width in them. + * [17:16] load format: 0 F32Z, 1 I24ZI8S, 2 I16ZI16Z, 3 F32ZI8S1V. + * [19:18] store format, the same encoding: 0 stores 0xbe800000, + * 1 stores 0x00400000, 2 stores 0xc000c000, 3 renders nothing. + * The vendor's 0x05/0x0a/0x0f are the pairs (1,1), (2,2), (3,3). + * bit 20 ZONLYRENDER, bit 21 LOADMASK, bit 22 STOREMASK. + * + * The enable is two-level, which is what the single-bit sweep could not see: + * the master bit honours the per-region ZLOAD/ZSTOREENABLE header bits the + * tiler leaves clear, and the FORCE bit overrides them for every region - + * so ZLOADEN alone loaded nothing and FORCEZLOAD alone loaded nothing, and + * the working store was ZSTOREEN+FORCEZSTORE all along, not two mystery + * enables. A continuation pass ORs ZLOADEN|SLOADEN|FORCEZLOAD, which is what + * the vendor's GL does for the pass after a mid-scene render; the load and + * store bases at 0x0484-0x0490 already carry the depth buffer on every pass. + * FORCEZLOAD reads the flat ZLOAD_BASE with the ISP addressing each tile the + * mirror of how FORCEZSTORE already writes - the per-region header path was a + * dead end, because the tiler leaves the region ZLS base (dword 2) zero. + * + * The one further thing a load pass needs is the background object's mask + * plane off (below): with it on, the object reset the loaded depth. With that + * cleared the split and splitblend depth oracle cases - the shape of + * ioquake3's GL_EQUAL dynamic-light pass - render correctly on hardware, + * against the far plane before. SGX_ZLS_AND, SGX_ZLS_OR and SGX_ZLS_SET + * rewrite the stream for experiments; SGX_ZLS_ALL applies the rewrite to a + * clearing pass too, and SGX_ZLS_BG replaces the background depth at 0x4b8. */ +void xpsb_ta_raster_set_z_load(uint32_t *p, unsigned n, int on) +{ + const char *ea = XPSB_ENVS("SGX_ZLS_AND"), *eo = XPSB_ENVS("SGX_ZLS_OR"); + uint32_t and_m = ea ? (uint32_t)strtoul(ea, NULL, 0) : XPSB_ZLS_LOAD_AND; + uint32_t or_m = eo ? (uint32_t)strtoul(eo, NULL, 0) : XPSB_ZLS_LOAD_OR; + unsigned i; + + if (!p || (!on && !XPSB_ENVS("SGX_ZLS_ALL"))) + return; + for (i = 0; i + 1 < n; i += 2) { + if (p[i] == 0x0480) { + p[i + 1] = (p[i + 1] & and_m) | or_m; + if (XPSB_ENVS("SGX_DEBUG")) + fprintf(stderr, "sgx: z load: 0x0480 = " + "%08x\n", p[i + 1]); + } + /* The background object's mask plane (EUR_CR_ISP_BGOBJ bit 8) + * is for an SPM partial render, and on a load pass it resets + * the tile depth the ZLS just loaded - the loaded near quad + * stopped occluding the far one. Cleared here so the object + * still runs its load-back (bit 9, ENABLEBGTAG, stays) without + * touching depth. Measured: with it set the split oracle shows + * the far plane, without it every depth case renders. */ + else if (on && p[i] == 0x04bc) + p[i + 1] &= ~XPSB_BGOBJ_MASK; + /* The background object's depth, per pass: whether it writes + * the tile depth here is what decides if a load could survive + * it. */ + else if (p[i] == 0x04b8 && XPSB_ENVS("SGX_ZLS_BG")) + p[i + 1] = (uint32_t)strtoul(XPSB_ENVS("SGX_ZLS_BG"), + NULL, 0); + } + /* SGX_ZLS_SET=reg:val[,reg:val] is SGX_RASTER_SET restricted to this + * pass, so a candidate can be tried on the continuation without also + * changing the pass that clears. */ + { + const char *e = XPSB_ENVS("SGX_ZLS_SET"); + + while (e && *e) { + unsigned long reg = strtoul(e, (char **)&e, 0), val; + + if (*e != ':') + break; + val = strtoul(e + 1, (char **)&e, 0); + for (i = 0; i + 1 < n; i += 2) + if (p[i] == (uint32_t)reg) + p[i + 1] = (uint32_t)val; + if (*e == ',') + e++; + } + } +} + +unsigned xpsb_gen_ta_raster_stream_coeffs(uint32_t *p, int w, int h, + const uint32_t *sgx_coeffs) +{ + unsigned n = xpsb_gen_ta_raster_stream(p, w, h); + + if (!sgx_coeffs) + return n; + return n + xpsb_vidshader_emit_coeff_regs(p + n, sgx_coeffs); +} + +/* The three ISP background-object registers, decoded in + * gl-re/scene-management.md section 5 and confirmed on hardware here: + * + * 0x4c4 background object pointer, ((gpu + pre_add) >> 4) | (ncoord << 25). + * Relocated against buffer 3 at pre_add 0x50 - rastgeom record 1, the + * load-back object - where the ordinary raster stream uses record 0, + * the clear. Pointing this back at record 0 makes a recovered frame + * lose everything the partial render drew, which is what it is for. + * 0x4bc low 8 bits the background stencil value, bits 9:8 control. The low + * byte provably does nothing without a stencil test: 0x2ff and 0x200 + * render identically. + * 0x4b8 background depth, an IEEE-754 float. Rewriting it with the same + * bits is byte-identical; 0.0f instead of 1.0f loses 171110 of 249920 + * covered pixels, everything failing the depth test against a near + * background. + * + * Bit 8 of 0x4bc is why the recovery used to lose the tail of a flushed macro + * tile. psb_dri.so writes stencil|0x300 in the ordinary raster chunk and + * stencil|0x200 here, and 0x200 is what was captured - but on this frame the + * bit has to stay set, or every recovery drops geometry. With it set, three + * scenes over three heap sizes all come back byte-identical to the same scene + * rendered without a recovery; with it clear, all of them lose. + * + * Bit 8 is EUR_CR_ISP_BGOBJ_MASK (sgx535defs.h:1427-1428), the background + * object's participation in the mask plane, which is what a partial render + * builds. That is why a recovery needs it set and a plain depth load needs it + * clear: the first is accumulating a plane, the second has none. */ +unsigned xpsb_gen_oom_stream(uint32_t *p) +{ + static const uint32_t s[] = { + 0x04c4, 0x02008005, /* relocated */ + 0x04bc, 0x00000300, + 0x04b8, 0x3f800000, + }; + memcpy(p, s, sizeof s); + return sizeof s / 4; +} + +void xpsb_ta_raster_set_bg(uint32_t *p, unsigned n, uint32_t depth, + unsigned stencil) +{ + unsigned i; + + if (!p) + return; + for (i = 0; i + 1 < n; i += 2) { + if (p[i] == 0x04b8) + p[i + 1] = depth; + else if (p[i] == 0x04bc) + p[i + 1] = (p[i + 1] & ~0xffu) | (stencil & 0xffu); + } +} + +void xpsb_rastgeom_set_z(uint32_t *b, unsigned first, uint32_t z) +{ + unsigned r; + + if (!b) + return; + for (r = first; r < 2; r++) { + uint32_t *p = b + rast_rec[r].vtx; + + p[1] = z; p[3] = z; p[5] = z; + } +} + +/* The present blit's destination is a rectangle of 16-pixel screen tiles, + * carried by raster registers 0x40c (first tile) and 0x410 (last tile) - not + * by anything in the heap. Confirmed by capturing the binary stack at two + * window sizes and diffing. */ +unsigned xpsb_gen_raster_stream(uint32_t *p, int w, int h) +{ + uint32_t s[] = { + 0x0480, 0x00000000, 0x04c4, 0x02008100, /* relocated */ + 0x04bc, 0x00000300, 0x04dc, 0x00000000, + 0x04b8, 0x3f800000, 0x0404, 0x00000003, + 0x0414, 0x00000100, 0x041c, 0x00000000, + 0x0420, 0x00000000, 0x04c8, 0x00000088, + 0x042c, 0x00000000, 0x0a5c, 0x20050020, /* relocated */ + 0x0a60, 0x00000004, 0x0a64, 0x00004fff, + 0x040c, 0x00000000, 0x0410, 0x00000000, /* tile range, below */ + }; + s[29] = ((uint32_t)(XPSB_PRESENT_X / 16) << 16) | + (uint32_t)(XPSB_PRESENT_Y / 16); + s[31] = ((uint32_t)((XPSB_PRESENT_X + w - 1) / 16) << 16) | + (uint32_t)((XPSB_PRESENT_Y + h - 1) / 16); + memcpy(p, s, sizeof s); + return sizeof s / 4; +} + +/* ---- buffer objects ---- */ + +static int drm_call(int fd, unsigned long req, void *arg) +{ + int rc; + do { rc = ioctl(fd, req, arg); } while (rc == -1 && errno == EINTR); + return rc; +} + +static int bo_new(struct xpsb_frame *f, struct xpsb_bo *b, uint64_t size, + uint64_t create) +{ + union bo_create_arg c; + + memset(&c, 0, sizeof c); + c.req.size = size; + c.req.mask = create; + c.req.page_alignment = 1; + if (drm_call(f->fd, IOC_BO_CREATE, &c)) return -1; + b->handle = c.rep.handle; b->size = c.rep.size; b->gpu = c.rep.offset; + b->create = create; + b->token = c.rep.arg_handle; b->start = c.rep.buffer_start; + b->base = mmap64(NULL, (size_t)(b->size + b->start), PROT_READ | PROT_WRITE, + MAP_SHARED, f->fd, (off64_t)b->token); + if (b->base == MAP_FAILED) { b->base = NULL; return -1; } + b->cpu = (char *)b->base + b->start; + return 0; +} + +/* CPU access is bracketed by BO_MAP/BO_UNMAP separately from the mmap. */ +static int bo_begin(struct xpsb_frame *f, struct xpsb_bo *b) +{ + union bo_map_arg m; + + memset(&m, 0, sizeof m); + m.req.handle = b->handle; + m.req.mask = DRM_BO_FLAG_READ | DRM_BO_FLAG_WRITE; + if (drm_call(f->fd, IOC_BO_MAP, &m)) return -1; + b->gpu = m.rep.offset; + return 0; +} + +static void bo_end(struct xpsb_frame *f, struct xpsb_bo *b) +{ + struct bo_handle_arg h; + + h.handle = b->handle; + drm_call(f->fd, IOC_BO_UNMAP, &h); +} + +static void bo_free(struct xpsb_frame *f, struct xpsb_bo *b) +{ + struct bo_handle_arg h; + + if (b->base) munmap(b->base, (size_t)(b->size + b->start)); + if (b->handle) { + h.handle = b->handle; + drm_call(f->fd, IOC_BO_UNREFERENCE, &h); + } + memset(b, 0, sizeof *b); +} + +int xpsb_frame_map(struct xpsb_frame *f, unsigned idx) +{ + if (idx > XPSB_FRAME_SUBST) { errno = EINVAL; return -1; } + return bo_begin(f, &f->bo[idx]); +} + +void xpsb_frame_unmap(struct xpsb_frame *f, unsigned idx) +{ + if (idx <= XPSB_FRAME_SUBST) bo_end(f, &f->bo[idx]); +} + +static void build_list(struct xpsb_frame *f, struct bo_op_arg *v, const int *idx, + unsigned n, const uint64_t *vf, const uint64_t *vm) +{ + unsigned i; + + for (i = 0; i < n; i++) { + int b = idx[i] < 0 ? XPSB_FRAME_SUBST : idx[i]; + memset(&v[i], 0, sizeof v[0]); + v[i].d.req.op = 0; + v[i].d.req.bo_req.handle = idx[i] < 0 ? f->bo[b].handle + : f->vhandle[b]; + v[i].d.req.bo_req.flags = vf ? vf[i] : xpsb_frame_bufs[idx[i]].vflags; + v[i].d.req.bo_req.mask = vm ? vm[i] : xpsb_frame_bufs[idx[i]].vmask; + v[i].next = (i + 1 < n) ? (uint64_t)(uintptr_t)&v[i + 1] : 0; + } +} + +static int write_relocs(struct xpsb_frame *f) +{ + struct xpsb_reloc ta[XPSB_MAX_TA_RELOCS], ras[XPSB_NUM_RAS_RELOCS]; + unsigned i, n, patched = 0; + + n = xpsb_gen_ta_relocs(ta, &f->rcfg); + if (!n) { errno = EINVAL; return -1; } + /* heap+0x3e4 has room for 14 indices, so retarget the one relocation + * that produces the index address into free space in buffer 5. */ + for (i = 0; i < n; i++) + if (ta[i].reloc_op == XPSB_RELOC_OP_OFFSET && + ta[i].buffer == XPSB_BUF_HEAP && ta[i].pre_add == 0x3e4 && + ta[i].dst_buffer == XPSB_BUF_VTX) { + ta[i].buffer = XPSB_BUF_VTX; + ta[i].pre_add = XPSB_IDX_OFF; + patched++; + } + if (patched != 1) { errno = EINVAL; return -1; } + f->nta_reloc = n; + f->nras_reloc = xpsb_gen_raster_relocs(ras, &f->rcfg); + + if (bo_begin(f, &f->relbo)) return -1; + memcpy((char *)f->relbo.cpu + XPSB_TA_RELOC_OFF, ta, n * sizeof ta[0]); + memcpy((char *)f->relbo.cpu + XPSB_RAS_RELOC_OFF, ras, + f->nras_reloc * sizeof ras[0]); + bo_end(f, &f->relbo); + return 0; +} + +/* Writes the secondary PDS program and points the binding at it. It leaves + * heap+0x380 either because it carries constants and needs a data segment, or + * because a second sampled unit has pushed the primary code segment onto that + * address; both land in the same free block. + * + * The video path's eleven conversion floats and a composite's single scalar are + * the same emitter's two cases and the same bank, so they cannot both be on. */ +static int frame_write_sec_pds(struct xpsb_frame *f, uint32_t *h) +{ + uint32_t prog[XPSB_SEC_PDS_MAX], off, const_off = 0; + unsigned n = 0, dsize = 0, nattr = 0; + + if (f->video_on) { + const_off = XPSB_VID_CONST_OFF; + nattr = XPSB_VID_CONST_N; + n = xpsb_gen_sec_pds(prog, f->video_conv, XPSB_VID_CONST_N, + XPSB_HEAP_ADDR + const_off, &dsize); + } else if (f->scalar_on) { + nattr = 1; + n = xpsb_gen_sec_pds(prog, (const float *)&f->scalar, 1, 0, + &dsize); + } + if (nattr && !n) { errno = EINVAL; return -1; } + off = (n || f->rcfg.ntex > 1) ? XPSB_SEC_PDS_OFF : 0; + + memset(h + XPSB_SEC_PDS_OFF / 4, 0, XPSB_SEC_PDS_MAX * 4); + if (n) + memcpy(h + XPSB_SEC_PDS_OFF / 4, prog, n * 4); + else if (off) + h[XPSB_SEC_PDS_OFF / 4] = XPSB_PDS_HALT; + + xpsb_heap_set_sec_pds(h, off, dsize, nattr); + f->rcfg.sec_pds_off = off; + f->rcfg.sec_pds_dwords = dsize; + f->rcfg.sec_const_off = const_off; + return 0; +} + +int xpsb_frame_set_geometry(struct xpsb_frame *f, int w, int h, + uint32_t stride, int depth_test) +{ + uint32_t sp = stride ? stride : (((uint32_t)w + 31u) & ~31u); + + /* One pixel, not one tile: the render box is counted in tiles and a + * target smaller than one still fills one. See the note in + * xpsb_ta_raster_set_msaa(). */ + if (w < 1 || h < 1 || sp < (uint32_t)w) { errno = EINVAL; return -1; } + /* EURASIA_RENDERSIZE_MAXX/MAXY - refused rather than clamped, the way + * SGXAddRenderTarget() refuses it. */ + if ((unsigned)w > XPSB_RENDER_SIZE_MAX || + (unsigned)h > XPSB_RENDER_SIZE_MAX) { errno = EINVAL; return -1; } + if ((uint64_t)sp * (uint32_t)h * 4 > f->bo[XPSB_BUF_RT].size) { + errno = ENOSPC; + return -1; + } + + f->w = w; f->h = h; f->stride = sp; f->depth_test = depth_test; + f->scene.w = (uint32_t)w; f->scene.h = (uint32_t)h; + f->dest.handle = f->bo[XPSB_BUF_RT].handle; + f->dest.offset = 0; + f->dest.w = (uint32_t)w; f->dest.h = (uint32_t)h; + f->dest.stride = sp * 4; + f->dest.format = XPSB_FMT_8888; + f->vhandle[XPSB_BUF_RT] = f->bo[XPSB_BUF_RT].handle; + + if (xpsb_frame_map(f, XPSB_BUF_HEAP)) return -1; + xpsb_gen_heap(f->bo[XPSB_BUF_HEAP].cpu, w, h, sp, depth_test); + /* the heap is regenerated from scratch, so re-apply the state that does + * not come from the geometry */ + xpsb_heap_set_blend(f->bo[XPSB_BUF_HEAP].cpu, f->blend); + xpsb_heap_set_ntex(f->bo[XPSB_BUF_HEAP].cpu, f->rcfg.ntex); + xpsb_heap_set_video(f->bo[XPSB_BUF_HEAP].cpu, + f->video_on ? f->video_conv : NULL); + xpsb_heap_set_frag_use(f->bo[XPSB_BUF_HEAP].cpu, f->rcfg.frag_use_off); + if (frame_write_sec_pds(f, f->bo[XPSB_BUF_HEAP].cpu)) { + xpsb_frame_unmap(f, XPSB_BUF_HEAP); + return -1; + } + xpsb_frame_unmap(f, XPSB_BUF_HEAP); + + if (xpsb_frame_map(f, XPSB_BUF_RASTGEOM)) return -1; + xpsb_gen_rastgeom_from(f->bo[XPSB_BUF_RASTGEOM].cpu, w, h, + f->rcfg.first_draw); + xpsb_frame_unmap(f, XPSB_BUF_RASTGEOM); + + if (xpsb_frame_map(f, XPSB_BUF_VTX)) return -1; + f->ndraw = xpsb_gen_draw_records_n(f->bo[XPSB_BUF_VTX].cpu, + f->rcfg.first_draw, f->rcfg.ntex); + f->draw_cmd_off = xpsb_draw_cmd_off(f->rcfg.first_draw); + xpsb_frame_unmap(f, XPSB_BUF_VTX); + + if (xpsb_frame_map(f, XPSB_BUF_CMD)) return -1; + { + char *c = f->bo[XPSB_BUF_CMD].cpu; + f->nta = xpsb_gen_ta_stream((uint32_t *)(c + XPSB_TA_TA_OFF)); + f->ncmd = xpsb_gen_ta_raster_stream((uint32_t *)(c + XPSB_TA_CMD_OFF), w, h); + f->noom = xpsb_gen_oom_stream((uint32_t *)(c + XPSB_TA_OOM_OFF)); + f->nras = xpsb_gen_raster_stream((uint32_t *)(c + XPSB_RAS_CMD_OFF), w, h); + } + xpsb_frame_unmap(f, XPSB_BUF_CMD); + return 0; +} + +int xpsb_frame_init(struct xpsb_frame *f, int drmfd, int w, int h) +{ + uint32_t sp = ((uint32_t)w + 31u) & ~31u; + int err; + unsigned i; + + memset(f, 0, sizeof *f); + f->fd = drmfd; + f->scene.num_buffers = XPSB_SCENE_NB; + f->rcfg.first_draw = 0; + f->rcfg.ntex = 1; + f->texctl_mode = XPSB_TEXCTL_CAPTURE; + f->blend_op = -1; + + for (i = 0; i < XPSB_FRAME_NBUF; i++) { + uint64_t sz = xpsb_frame_bufs[i].size; + uint64_t need = (uint64_t)sp * (uint32_t)h * 4; + + if ((i == XPSB_BUF_RT || i == XPSB_BUF_DEPTH) && sz < need) sz = need; + if (i == XPSB_BUF_TEX && sz < XPSB_TEX_BYTES) sz = XPSB_TEX_BYTES; + if (bo_new(f, &f->bo[i], sz, xpsb_frame_bufs[i].create)) goto fail; + f->vhandle[i] = f->bo[i].handle; + } + /* The raster pass writes into a buffer the X server owns; until the + * caller supplies one, a private BO stands in. */ + if (bo_new(f, &f->bo[XPSB_FRAME_SUBST], 4u << 20, + DRM_BO_FLAG_MEM_TT | DRM_BO_FLAG_READ | DRM_BO_FLAG_WRITE | + DRM_BO_FLAG_MAPPABLE)) goto fail; + if (bo_new(f, &f->relbo, 1u << 20, + DRM_BO_FLAG_MEM_LOCAL | DRM_BO_FLAG_READ | DRM_BO_FLAG_WRITE | + DRM_BO_FLAG_MAPPABLE)) goto fail; + if (xpsb_frame_map(f, XPSB_BUF_USSE)) goto fail; + xpsb_gen_usse(f->bo[XPSB_BUF_USSE].cpu, XPSB_CLEAR_DEFAULT); + xpsb_frame_unmap(f, XPSB_BUF_USSE); + if (xpsb_frame_set_geometry(f, w, h, 0, 1)) goto fail; + if (write_relocs(f)) goto fail; + return 0; +fail: + err = errno; + xpsb_frame_fini(f); + errno = err; + return -1; +} + +void xpsb_frame_fini(struct xpsb_frame *f) +{ + unsigned i; + + for (i = 0; i <= XPSB_FRAME_SUBST; i++) bo_free(f, &f->bo[i]); + bo_free(f, &f->relbo); +} + +/* ---- caller-supplied surfaces, quad and clear ---- */ + +int xpsb_frame_set_dest(struct xpsb_frame *f, const struct xpsb_surface_desc *d) +{ + unsigned bpp = xpsb_format_bpp(d->format); + + if (xpsb_surface_check(d)) return -1; + if (xpsb_frame_map(f, XPSB_BUF_HEAP)) return -1; + if (xpsb_heap_set_dest(f->bo[XPSB_BUF_HEAP].cpu, d)) { + int e = errno; + xpsb_frame_unmap(f, XPSB_BUF_HEAP); + errno = e; + return -1; + } + xpsb_frame_unmap(f, XPSB_BUF_HEAP); + + f->dest = *d; + f->w = (int)d->w; f->h = (int)d->h; + f->stride = d->stride / bpp; + f->scene.w = d->w; f->scene.h = d->h; + f->vhandle[XPSB_BUF_RT] = d->handle; + f->rcfg.dest_offset = d->offset; + return write_relocs(f); +} + +int xpsb_frame_set_texctl_mode(struct xpsb_frame *f, int mode) +{ + if (mode != XPSB_TEXCTL_CAPTURE && mode != XPSB_TEXCTL_XPSB) { + errno = EINVAL; + return -1; + } + f->texctl_mode = mode; + return 0; +} + +/* Everything a second sampled unit needs, in the order the stream carries it: + * the descriptor at heap+0x350/0x354, the address slot and its relocation at + * where=0x370 against validate position 8, the primary PDS program grown to a + * 16-dword data segment and a six-dword code segment, the two binding words + * that declare that size and the extra iterated varying, the secondary program + * displaced out of heap+0x380, the vertex record 44 -> 56 bytes with its DMA + * control word, emit count and draw-record granule count following it. + * + * Three of those are derived rather than captured, and each says so where it is + * written: the two iterator control words' coordinate-set nibble, the vertex + * USE program's emit count, and the draw record's granule count. + * xpsb_3d.c holds XPSB_3D_CAP_MASK_TEX closed until they are confirmed on + * hardware; the frame layer itself no longer refuses the configuration. + * + * All three have since been confirmed, and a two-unit frame has rendered on + * hardware - tools/baremetal/sgxtri with SGXTRI_NTEX=2, written up in + * work/attrib-binding/. Primary attributes come out allocated strictly in + * sequence: pa0 the iterated colour, then pa[1 + u] the sampled texel of unit + * u, shown by giving unit 1 a flat texel unlike unit 0's checkerboard. + * + * That frame also found the trap in this path. At ntex == 2 the primary PDS + * program grows to a sixteen-dword data segment, so its code segment starts at + * heap+0x380 - exactly where the secondary program's bare HALT sits. + * xpsb_heap_set_ntex() rebuilds the program but does not move that HALT, which + * is why xpsb_frame_set_ntex() calls frame_write_sec_pds() right after it. + * Anyone driving the heap setter directly must displace the secondary program + * too, or the frame does not fault - it times out with the watchdog still + * seeing motion. */ +int xpsb_frame_set_ntex(struct xpsb_frame *f, unsigned n) +{ + if (n < 1 || n > XPSB_NTEX_MAX) { errno = EINVAL; return -1; } + if (f->rcfg.ntex == n) return 0; + f->rcfg.ntex = n; + + if (xpsb_frame_map(f, XPSB_BUF_HEAP)) return -1; + xpsb_heap_set_ntex(f->bo[XPSB_BUF_HEAP].cpu, n); + if (frame_write_sec_pds(f, f->bo[XPSB_BUF_HEAP].cpu)) { + xpsb_frame_unmap(f, XPSB_BUF_HEAP); + return -1; + } + xpsb_frame_unmap(f, XPSB_BUF_HEAP); + + if (xpsb_frame_map(f, XPSB_BUF_USSE)) return -1; + xpsb_usse_set_vtx_dwords(f->bo[XPSB_BUF_USSE].cpu, xpsb_vtx_stride(n)); + xpsb_frame_unmap(f, XPSB_BUF_USSE); + + if (xpsb_frame_map(f, XPSB_BUF_VTX)) return -1; + f->ndraw = xpsb_gen_draw_records_n(f->bo[XPSB_BUF_VTX].cpu, + f->rcfg.first_draw, n); + xpsb_frame_unmap(f, XPSB_BUF_VTX); + return write_relocs(f); +} + +int xpsb_frame_set_texture(struct xpsb_frame *f, int unit, + const struct xpsb_surface_desc *t) +{ + uint32_t w0, w1, *h; + + if (unit < 0 || unit >= (int)f->rcfg.ntex) { errno = EINVAL; return -1; } + if (xpsb_tex_state_mode(&w0, &w1, t, f->texctl_mode)) return -1; + + if (xpsb_frame_map(f, XPSB_BUF_HEAP)) return -1; + h = f->bo[XPSB_BUF_HEAP].cpu; + h[(XPSB_TEXCTL_OFF + unit * XPSB_TEX_UNIT_STEP) / 4] = w0; + h[(XPSB_TEXSTATE_OFF + unit * XPSB_TEX_UNIT_STEP) / 4] = w1; + xpsb_frame_unmap(f, XPSB_BUF_HEAP); + + f->tex[unit] = *t; + f->vhandle[XPSB_BUF_TEX + unit] = t->handle; + f->rcfg.tex_offset[unit] = t->offset; + return write_relocs(f); +} + +int xpsb_frame_set_quad(struct xpsb_frame *f, const struct xpsb_quad *q) +{ + float vtx[4 * XPSB_VTX_STRIDE_MAX]; + size_t bytes = 4 * xpsb_vtx_stride(f->rcfg.ntex) * sizeof vtx[0]; + uint16_t idx[6]; + char *b; + + f->nidx = xpsb_gen_quad_n(vtx, idx, q, f->rcfg.ntex, NULL); + if (!f->nidx) { errno = EINVAL; return -1; } + if (xpsb_frame_map(f, XPSB_BUF_VTX)) return -1; + b = f->bo[XPSB_BUF_VTX].cpu; + memcpy(b + XPSB_VTX_OFF, vtx, bytes); + memcpy(b + XPSB_IDX_OFF, idx, sizeof idx); + *(uint32_t *)(b + f->draw_cmd_off) = XPSB_DRAW_CMD_TAG | f->nidx; + xpsb_frame_unmap(f, XPSB_BUF_VTX); + return 0; +} + +/* Dropping the clear changes the draw-record array, the rastgeom array and ten + * relocations at once. Nothing in test/ has ever submitted that stream, so the + * clear stays on until a readback on hardware shows the destination's prior + * contents surviving outside the quad. */ +int xpsb_frame_set_clear(struct xpsb_frame *f, int on) +{ + unsigned want = on ? 0 : XPSB_USER_DRAW; + + if (f->rcfg.first_draw == want) return 0; + f->rcfg.first_draw = want; + + if (xpsb_frame_map(f, XPSB_BUF_RASTGEOM)) return -1; + memset(f->bo[XPSB_BUF_RASTGEOM].cpu, 0, 0x400 * 4); + xpsb_gen_rastgeom_from(f->bo[XPSB_BUF_RASTGEOM].cpu, f->w, f->h, want); + xpsb_frame_unmap(f, XPSB_BUF_RASTGEOM); + + if (xpsb_frame_map(f, XPSB_BUF_VTX)) return -1; + memset(f->bo[XPSB_BUF_VTX].cpu, 0, + (XPSB_NUM_DRAW + 1) * XPSB_DRAW_STRIDE * 4); + f->ndraw = xpsb_gen_draw_records_n(f->bo[XPSB_BUF_VTX].cpu, want, + f->rcfg.ntex); + f->draw_cmd_off = xpsb_draw_cmd_off(want); + xpsb_frame_unmap(f, XPSB_BUF_VTX); + + return write_relocs(f); +} + +/* Blending touches exactly two places: enable bit 25 of the per-draw ISP word, + * and the fragment slot, which carries the Porter-Duff factors as the two SOP2 + * instructions xpsb_composite_shader() emits. Nothing else in the heap moves - + * captures that differ only in blend factors differ by no heap dwords at all. + * + * Never submitted on this hardware by our code, so it stays off by default and + * the caller has to ask for it by operator. */ +static int frame_write_frag(struct xpsb_frame *f, const uint32_t *slot) +{ + if (xpsb_frame_map(f, XPSB_BUF_USSE)) return -1; + memcpy((char *)f->bo[XPSB_BUF_USSE].cpu + XPSB_USSE_FRAG_OFF, slot, + XPSB_USSE_SLOT_DW * 4); + xpsb_frame_unmap(f, XPSB_BUF_USSE); + return 0; +} + +int xpsb_frame_set_composite(struct xpsb_frame *f, int op, + const struct xpsb_shader_flags *fl) +{ + uint32_t slot[XPSB_USSE_SLOT_DW]; + + if (xpsb_usse_composite(slot, op, fl)) return -1; + if (frame_write_frag(f, slot)) return -1; + /* Two instructions, which is what xpsb_composite_shader() builds. The + * reservation was never set here at all, so it kept the captured + * one-instruction size while the program behind it was twice that. */ + f->rcfg.frag_use_off = XPSB_USSE_FRAG_OFF; + f->rcfg.frag_use_size = XPSB_COMPOSITE_DW * 4u; + + if (xpsb_frame_map(f, XPSB_BUF_HEAP)) return -1; + xpsb_heap_set_blend(f->bo[XPSB_BUF_HEAP].cpu, 1); + xpsb_frame_unmap(f, XPSB_BUF_HEAP); + + f->blend = 1; + f->blend_op = op; + f->blend_flags = *fl; + return 0; +} + +/* The constant reaches sa0 as PDS data, so it is the program itself that + * changes when it does; the binding then declares one secondary attribute so + * the bank the DOUTA writes into is allocated. */ +int xpsb_frame_set_scalar(struct xpsb_frame *f, int on, uint32_t argb) +{ + int prev = f->scalar_on, ret; + + if (on && f->video_on) { errno = EINVAL; return -1; } + /* the constant is copied as four bytes, never read as a float */ + f->scalar_on = on ? 1 : 0; + f->scalar = argb; + + if (xpsb_frame_map(f, XPSB_BUF_HEAP)) return -1; + ret = frame_write_sec_pds(f, f->bo[XPSB_BUF_HEAP].cpu); + xpsb_frame_unmap(f, XPSB_BUF_HEAP); + if (ret) { + f->scalar_on = prev; + return -1; + } + return write_relocs(f); +} + +/* Everything the packed YUV path adds, in the order the stream carries it: the + * FOURCC's program into the video USSE slot with its texel source moved to pa1, + * the temporary-register count and the eleven floats in the heap, the secondary + * PDS program that DMAs them into sa0..sa10, and the user draw's USE task + * pointed at the video slot instead of the fragment one. + * + * Unchanged, deliberately: the DOUTI iterator word at heap 0xd9, the USE task's + * 0x180000 dependency background, the eleven-float vertex record and the one + * sampled unit. Route B keeps every plumbing dword that has already submitted + * and moves the shader instead (disasm/video-integration.md sections 1 and 2). + * + * The temporary-register count is the one word here that no capture of this + * plumbing carries; see XPSB_VID_TEMPS. */ +int xpsb_frame_set_video(struct xpsb_frame *f, uint32_t fourcc, + const float *conv) +{ + int prev = f->video_on, ret; + unsigned vid_dw = 0; + uint32_t *h; + + if (!fourcc && !f->video_on) + return 0; + if (fourcc) { + if (xpsb_vidshader_offset(fourcc) != XPSB_USSE_OFF_PACKED || + !conv || f->scalar_on) { + errno = f->scalar_on ? EINVAL : ENOSYS; + return -1; + } + } + f->video_on = fourcc ? 1 : 0; + f->video_fourcc = fourcc; + if (f->video_on) + memcpy(f->video_conv, conv, sizeof f->video_conv); + + if (f->video_on) { + if (xpsb_frame_map(f, XPSB_BUF_USSE)) goto fail; + vid_dw = xpsb_usse_video(f->bo[XPSB_BUF_USSE].cpu, fourcc); + ret = !vid_dw || + xpsb_usse_video_pa(f->bo[XPSB_BUF_USSE].cpu, fourcc, + XPSB_VID_SRC_PA); + xpsb_frame_unmap(f, XPSB_BUF_USSE); + if (ret) { errno = EINVAL; goto fail; } + } + + if (xpsb_frame_map(f, XPSB_BUF_HEAP)) goto fail; + h = f->bo[XPSB_BUF_HEAP].cpu; + xpsb_heap_set_video(h, f->video_on ? f->video_conv : NULL); + xpsb_heap_set_frag_use(h, f->video_on ? XPSB_USSE_VID_OFF : 0); + ret = frame_write_sec_pds(f, h); + xpsb_frame_unmap(f, XPSB_BUF_HEAP); + if (ret) goto fail; + + f->rcfg.frag_use_off = f->video_on ? XPSB_USSE_VID_OFF : 0; + /* What the emitter wrote, which it reports, rather than the whole + * slot: the reservation names the program the USE register has to + * cover, and reserving the slot claimed more of the window than the + * program occupies. */ + f->rcfg.frag_use_size = f->video_on ? vid_dw * 4u : 0; + return write_relocs(f); +fail: + ret = errno; + f->video_on = prev; + errno = ret; + return -1; +} + +int xpsb_frame_set_blend(struct xpsb_frame *f, int on) +{ + uint32_t slot[XPSB_USSE_SLOT_DW]; + + if (on) { + if (f->blend_op < 0) { errno = EINVAL; return -1; } + return xpsb_frame_set_composite(f, f->blend_op, &f->blend_flags); + } + + xpsb_usse_default_frag(slot); + if (frame_write_frag(f, slot)) return -1; + + if (xpsb_frame_map(f, XPSB_BUF_HEAP)) return -1; + xpsb_heap_set_blend(f->bo[XPSB_BUF_HEAP].cpu, 0); + xpsb_frame_unmap(f, XPSB_BUF_HEAP); + + f->blend = 0; + return 0; +} + +int xpsb_frame_submit(struct xpsb_frame *f) +{ + struct bo_op_arg vlist[XPSB_FRAME_NBUF + XPSB_NTEX_MAX - 1]; + int talist[XPSB_FRAME_NBUF + XPSB_NTEX_MAX - 1]; + uint64_t tavf[XPSB_FRAME_NBUF + XPSB_NTEX_MAX - 1]; + uint64_t tavm[XPSB_FRAME_NBUF + XPSB_NTEX_MAX - 1]; + unsigned nlist = XPSB_FRAME_NBUF + f->rcfg.ntex - 1, i; + struct psb_cmdbuf_arg ca; + struct fence_arg fa; + + /* Sampled units past the first extend the TA list, so their relocations + * can name validate position 8 upwards; they place like buffer 7. */ + for (i = 0; i < nlist; i++) { + talist[i] = i < XPSB_FRAME_NBUF ? xpsb_ta_list[i] : (int)i; + tavf[i] = i < XPSB_FRAME_NBUF ? xpsb_frame_bufs[i].vflags : 0; + tavm[i] = i < XPSB_FRAME_NBUF ? xpsb_frame_bufs[i].vmask : 0; + } + build_list(f, vlist, talist, nlist, tavf, tavm); + memset(&fa, 0, sizeof fa); memset(&ca, 0, sizeof ca); + ca.engine = XPSB_TA_ENGINE; ca.ta_flags = XPSB_TA_TAFLAGS; + ca.cmdbuf_handle = f->bo[XPSB_BUF_CMD].handle; + ca.cmdbuf_offset = XPSB_TA_CMD_OFF; ca.cmdbuf_size = f->ncmd; + ca.ta_handle = f->bo[XPSB_BUF_CMD].handle; + ca.ta_offset = XPSB_TA_TA_OFF; ca.ta_size = f->nta; + ca.oom_handle = f->bo[XPSB_BUF_CMD].handle; + ca.oom_offset = XPSB_TA_OOM_OFF; ca.oom_size = f->noom; + ca.reloc_handle = f->relbo.handle; ca.reloc_offset = XPSB_TA_RELOC_OFF; + ca.num_relocs = f->nta_reloc; + ca.buffer_list = (uint64_t)(uintptr_t)&vlist[0]; + ca.scene_arg = (uint64_t)(uintptr_t)&f->scene; + ca.fence_arg = (uint64_t)(uintptr_t)&fa; + if (drm_call(f->fd, IOC_PSB_CMDBUF, &ca)) return -1; + if (fa.handle && fa.handle != ~0u) { + drm_call(f->fd, IOC_FENCE_WAIT, &fa); + drm_call(f->fd, IOC_FENCE_UNREF, &fa); + } + + build_list(f, vlist, xpsb_raster_list, XPSB_RAS_LIST_LEN, + xpsb_raster_vflags, xpsb_raster_vmask); + memset(&fa, 0, sizeof fa); memset(&ca, 0, sizeof ca); + ca.engine = XPSB_RAS_ENGINE; ca.ta_flags = XPSB_RAS_TAFLAGS; + ca.cmdbuf_handle = f->bo[XPSB_BUF_CMD].handle; + ca.cmdbuf_offset = XPSB_RAS_CMD_OFF; ca.cmdbuf_size = f->nras; + ca.ta_handle = f->bo[XPSB_BUF_CMD].handle; + ca.ta_offset = XPSB_RAS_TA_OFF; ca.ta_size = XPSB_RAS_TA_SIZE; + ca.oom_handle = f->bo[XPSB_BUF_CMD].handle; + ca.oom_offset = XPSB_RAS_OOM_OFF; ca.oom_size = XPSB_RAS_OOM_SIZE; + ca.reloc_handle = f->relbo.handle; ca.reloc_offset = XPSB_RAS_RELOC_OFF; + ca.num_relocs = f->nras_reloc; + ca.buffer_list = (uint64_t)(uintptr_t)&vlist[0]; + ca.fence_arg = (uint64_t)(uintptr_t)&fa; + if (drm_call(f->fd, IOC_PSB_CMDBUF, &ca)) return -1; + if (fa.handle && fa.handle != ~0u) { + drm_call(f->fd, IOC_FENCE_WAIT, &fa); + drm_call(f->fd, IOC_FENCE_UNREF, &fa); + } + return 0; +} + +int xpsb_frame_readback(struct xpsb_frame *f, void *dst, uint32_t dst_stride) +{ + const char *src; + char *d = dst; + uint32_t row = (uint32_t)f->w * 4, src_stride = f->stride * 4; + int y; + + if (!dst_stride) dst_stride = row; + if (f->vhandle[XPSB_BUF_RT] != f->bo[XPSB_BUF_RT].handle) { + errno = EINVAL; /* a caller-owned destination */ + return -1; + } + if (xpsb_frame_map(f, XPSB_BUF_RT)) return -1; + /* GPU memory is uncached, so pull whole rows rather than pixels: at + * 512x512 that is 0.7 s of user time over 60 frames against 7.9 s. */ + src = f->bo[XPSB_BUF_RT].cpu; + for (y = 0; y < f->h; y++) + memcpy(d + (size_t)y * dst_stride, + src + (size_t)y * src_stride, row); + xpsb_frame_unmap(f, XPSB_BUF_RT); + return 0; +} diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/xpsb_frame.h mesa-26.2.2/src/gallium/drivers/sgx/xpsb_frame.h --- mesa-26.2.2.orig/src/gallium/drivers/sgx/xpsb_frame.h 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/xpsb_frame.h 2026-09-08 10:57:36.684713558 +0200 @@ -0,0 +1,1308 @@ +/* Open replacement for Xpsb.so - the SGX frame renderer. + * + * Extracted from test/drmcube.c, which renders a textured cube on this + * hardware today. The buffer set, the parameter heap, the rastgeom records, + * the draw records, the four register/value streams, the relocation lists and + * the two-pass submission are all the same words that program produces. + * Parameterised beyond it: the render-target geometry, a caller-supplied + * destination and texture, a single quad, and dropping the two clearing draws. + * + * The DRM ABI below is the poulsbo TTM and psb ioctl interface, reproduced so + * this module builds without libdrm-poulsbo in the include path. Failing + * entry points return -1 with errno left from the ioctl that failed. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#ifndef _XPSB_FRAME_H_ +#define _XPSB_FRAME_H_ + +#include +#include + +#include "xpsb_shader.h" +#include "xpsb_vidshader.h" +#include "xpsb_pds.h" + +#define XPSB_FRAME_NBUF 8 +#define XPSB_FRAME_SUBST XPSB_FRAME_NBUF /* stands in for the X-owned dest */ + +#define XPSB_BUF_HEAP 0 +#define XPSB_BUF_RT 1 +#define XPSB_BUF_USSE 2 +#define XPSB_BUF_RASTGEOM 3 +#define XPSB_BUF_CMD 4 +#define XPSB_BUF_VTX 5 +#define XPSB_BUF_DEPTH 6 +#define XPSB_BUF_TEX 7 + +/* Buffer 7 is 16 KB in the capture; drmcube grows it to hold its 64x384 + * texture atlas, and that is the largest texture a frame can carry. */ +#define XPSB_TEX_BYTES ((uint32_t)64 * 384 * 4) + +/* Vertex block and draw command, decoded by differential capture: 11 floats + * per vertex at XPSB_VTX_OFF in buffer 5, index count in word 2 of the third + * draw record. Indices are retargeted out of the heap into buffer 5, which + * has room; heap+0x3e4 holds only 14. */ +#define XPSB_VTX_OFF 0x45800 +#define XPSB_VTX_STRIDE 11 +/* Each sampled unit adds a three-float coordinate set, so the record is + * 8 + 3n floats: 44 bytes at one unit, 56 at two. */ +#define XPSB_NTEX_MAX 2 +#define XPSB_VTX_STRIDE_MAX (8 + 3 * XPSB_NTEX_MAX) +#define XPSB_DRAW_CMD_OFF 0x40 +#define XPSB_DRAW_CMD_TAG 0x81400000u + +/* The primitive type the VDM assembles from the index list, bits [29:26] of + * that command word - the DDK's EURASIA_VDM_TYPE. The captured tag above + * carries TRIS, which is zero, so a triangle stream is the tag unchanged. + * + * Only this field varies with the primitive: the index-presence bits beside + * it are written the same for every type (the vendor sets IDXPRES2, 3 and 45 + * unconditionally), so they are part of the tag rather than something to + * compute. What changes with the type is how many indices the VDM consumes - + * one for a point, two for a line, three for a triangle - and that it works + * out for itself. */ +#define XPSB_VDM_TYPE_SHIFT 26 +/* Index presence, [25:22]: the DDK's IDXPRES2, IDXPRES3, IDXPRES45 and + * IDXPRES67. The captured tag has 2 and 45 set, which is the triangle it + * was capturing. */ +#define XPSB_VDM_IDXPRES_MASK 0x03c00000u +#define XPSB_VDM_TRIS 0u +#define XPSB_VDM_LINES 1u +#define XPSB_VDM_POINTS 2u +#define XPSB_VDM_TYPE(t) ((uint32_t)(t) << XPSB_VDM_TYPE_SHIFT) +#define XPSB_DRAW_TERM_CMD 0xc0000000u /* word 2 of the terminator record */ + +/* Dword 4 of a draw record is the DDK's index list word 2 - the block is + * header, list 1, list 2, list 4, list 5, list 3 being absent because the + * captured tag leaves EURASIA_VDM_IDXPRES3 clear. [26:25] of it is + * EURASIA_VDM_FLATSHADE, the corner the vertex data master takes a flat value + * from: 0 vertex 0, 1 vertex 1, 2 vertex 2, 3 reserved + * (sgxdefs.h:1042-1047). There is no gouraud code - the field is the default + * VERTEX0 when nothing is flat, which is what + * EURASIA_VDM_WORD2_DEFAULT carries (sgxdefs.h:1056-1059) - so it is written + * only for a flat-shaded draw, as the vendor writes it + * (opengles1/validate.c:5013-5025). */ +#define XPSB_DRAW_IDX_WORD 4 +#define XPSB_VDM_FLAT_SHIFT 25 +#define XPSB_VDM_FLAT_MASK (3u << XPSB_VDM_FLAT_SHIFT) + +/* The index list word for a triangle corner named as 1 + the corner, or 0 to + * leave the record gouraud - the encoding struct xpsb_attribs::flatshade uses, + * so one value drives the MTE, the iterator and the vertex data master. */ +static inline uint32_t xpsb_vdm_idx_word(uint32_t w, unsigned flatshade) +{ + w &= ~XPSB_VDM_FLAT_MASK; + if (flatshade) + w |= ((flatshade - 1u) << XPSB_VDM_FLAT_SHIFT) & + XPSB_VDM_FLAT_MASK; + return w; +} + +#define XPSB_DRAW_STRIDE 7 /* dwords per draw record */ +#define XPSB_NUM_DRAW 3 /* real records, terminator excluded */ +#define XPSB_USER_DRAW 2 /* the one carrying caller geometry */ + +/* Dword 6 of a draw record counts the vertex record in four-dword granules, + * in bits [15:8] and again in bits [31:25]. Derived, not captured: the three + * records available are 4 dwords -> 0x02000103 and 11 -> 0x06000303 in the + * working stream and 8 -> 0x04000203 in GL-IMPLEMENTATION.md:217. */ +#define XPSB_DRAW_KIND(g) (0x3u | ((uint32_t)(g) << 8) | ((uint32_t)(g) << 25)) + +/* GPU address the captured heap sat at. Only ever a placeholder: every dword + * built from it is overwritten by a relocation. */ +#define XPSB_HEAP_ADDR 0x20010000u +#define XPSB_IDX_OFF 0x300000 +/* How many vertices of a given record width fit before the index buffer. + * + * The unparameterised form derived this from the captured eleven-float record + * while the live record is 8 + 3n floats, or the attribute stride - so a mesh + * that passed the test still ran its vertex records over the indices at two + * sampled units. Callers that know their stride should use the parameterised + * form; XPSB_MAX_VTX now assumes the widest record the frame can describe and + * is therefore safe for any of them. */ +#define XPSB_MAX_VTX_AT(stride) \ + ((XPSB_IDX_OFF - XPSB_VTX_OFF) / ((unsigned)(stride) * 4u)) +/* Eight fixed floats plus the widest coordinate sets the frame can carry. */ +#define XPSB_VTX_STRIDE_WIDEST (8 + XPSB_NSET_MAX * 4) +#define XPSB_MAX_VTX XPSB_MAX_VTX_AT(XPSB_VTX_STRIDE_WIDEST) + +/* Texture state words in the heap: unit 0's filter and wrap, then its format + * and size. They are the primary PDS program's own ds0[2] and ds0[3], so they + * move with it - use xpsb_texctl_dw() for a unit other than the first, whose + * dwords are not a fixed stride apart past the third. */ +#define XPSB_TEXCTL_OFF (XPSB_PRI_PDS_OFF + 8) +#define XPSB_TEXSTATE_OFF (XPSB_PRI_PDS_OFF + 0xc) + +#define XPSB_TEX_UNIT_STEP 8 +#define XPSB_MAX_TEX 8 + +/* Texture address list at heap+0x360: a constant, then (control, address) + * pairs at stride 8. Unit n's address is dword 0xda + 2n, which is where its + * relocation points. The vertex-fetch DMA descriptor is a separate block: + * dword 0xe9 is its control word and dword 0xf0 the vertex stride in bytes. + * + * The control word carries dwords - 1 twice, in bits [7:0] and bits [30:21]; + * every DOUTD descriptor in tools/isa-pds/pds_corpus.txt obeys that, and it is + * the same word xpsb_vidshader_emit_sa_pds() already builds. + * + * This form describes ONE line and is only valid up to sixteen dwords: the + * burst-size field is four bits, so seventeen and above wrap silently - + * twenty came out as burst 4 over two lines and the DMA delivered nothing + * usable (measured, see tools/baremetal/sgxtri.c). Use xpsb_dma_ctl() for a + * count that may exceed sixteen. */ +#define XPSB_TEXADDR_DW (XPSB_PRI_PDS_DW + 10) +#define XPSB_VTXDMA_DW 0xe9 +#define XPSB_VTXSTRIDE_DW 0xf0 +#define XPSB_DMA_CTL(n) (0x80000000u | ((uint32_t)((n) - 1) << 21) | \ + ((uint32_t)((n) - 1) & 0xffu)) + +/* A DOUTD control word for d dwords at attribute offset ao, splitting the + * transfer into lines when it is longer than one burst can carry. Returns 0 + * when no burst size divides it. */ +uint32_t xpsb_dma_ctl(unsigned d, unsigned ao); + +/* Describe the vertex record as nfloat floats wide, in every field that says + * so. */ +int xpsb_heap_set_record_width(uint32_t *h, unsigned nfloat, + int offset_colour); + +/* The per-draw MTE state block: a EURASIA_TACTLPRES_* presence mask (DDK + * sgxdefs.h:1240-1255) followed by the groups it names, in bit order. Nothing + * in it has a fixed position but the mask: a group's dword is the sum of the + * groups present below it, so every user goes through xpsb_heap_state_off(). + * + * The capture carries 0, 6, 8, 9, 10, 11 and 14 - fifteen dwords with the + * mask - at 0x1147, seventeen dwords short of the descriptor at 0x1158 that + * DMAs it. A block that outgrows that room moves past the descriptor to + * 0x1167, the last free run of the 0x100-byte window a record copies, and the + * descriptor's pointer follows it; xpsb_heap_state_base() says where it is + * from the descriptor's own dword count. */ +#define XPSB_HEAP_STATE 0x1147 /* the captured position */ +#define XPSB_HEAP_STATE_ALT 0x1167 /* where it goes when it outgrows */ +#define XPSB_HEAP_STATE_DESC 0x1158 /* address, then the DMA control */ +#define XPSB_HEAP_STATE_USE 0x115a /* the DOUTU of the copy program */ +#define XPSB_HEAP_STATE_END 0x1180 /* the window's end */ +#define XPSB_STATE_HOME_ROOM (XPSB_HEAP_STATE_DESC - XPSB_HEAP_STATE) +#define XPSB_STATE_ALT_ROOM (XPSB_HEAP_STATE_END - XPSB_HEAP_STATE_ALT) +#define XPSB_STATE_DWORDS 15 /* the captured block, mask included */ +#define XPSB_STATE_MASK_CAPTURED 0x00004f41u + +/* The groups, by presence bit. TERMINATE (13) carries no dwords: the DDK's + * EURASIA_TACTL_ALL_SIZEINDWORDS is 23 for this core (sgxdefs.h:1265), which + * is every other group's size summed. */ +#define XPSB_STATE_ISP_A 0 /* ISPCTLFF0: ISP word A */ +#define XPSB_STATE_ISP_B 1 /* ISPCTLFF1: word B, under BPRES */ +#define XPSB_STATE_ISP_C 2 /* ISPCTLFF2: word C, under CPRES */ +#define XPSB_STATE_ISP_BF_A 3 /* ISPCTLBF0-2: the back-face set, */ +#define XPSB_STATE_ISP_BF_B 4 /* all three, under word A 2SIDED */ +#define XPSB_STATE_ISP_BF_C 5 +#define XPSB_STATE_PDS 6 /* PDSSTATEPTR: three dwords */ +#define XPSB_STATE_RGNCLIP 7 /* two dwords */ +#define XPSB_STATE_VIEWPORT 8 /* six floats, centre then scale */ +#define XPSB_STATE_WRAP 9 +#define XPSB_STATE_OUTSEL 10 /* output selects, the G10 word */ +#define XPSB_STATE_WCLAMP 11 +#define XPSB_STATE_CULL 12 /* MTECTRL: cull, shade, clip mode */ +#define XPSB_STATE_TERMINATE 13 +#define XPSB_STATE_TEXSIZE 14 /* the coordinate-set word */ +#define XPSB_STATE_TEXFLOAT 15 +#define XPSB_STATE_NGROUPS 16 +#define XPSB_STATE_MAX_DWORDS 24 /* mask + EURASIA_TACTL_ALL_SIZEINDWORDS */ +#define XPSB_STATE_CULL_BIT (1u << XPSB_STATE_CULL) + +/* Dwords a group takes, or 0 for one the block does not know. */ +unsigned xpsb_state_size(unsigned group); +/* Dwords a block with this mask takes, the mask dword included. */ +unsigned xpsb_state_dwords(uint32_t mask); +/* Heap dword of the block's presence mask: home while a mask is there, else + * the alternate - a move zeroes the dwords it leaves, and a mask is never + * zero (ISP word A is always present). A block goes home while what the DMA + * carries for it fits before the descriptor: a seventeen-dword block is + * carried as eighteen, and the part did not take that from home (the first + * booklet run after the DMA fix), so it goes to the alternate like eighteen. */ +unsigned xpsb_heap_state_base(const uint32_t *h); +/* Dwords a DOUTD control word transfers: bursts times lines. */ +unsigned xpsb_dma_ctl_dwords(uint32_t ctl); +/* The primary attribute registers the state task is given, in the TA PDS + * state command's USEATTRIBUTESIZE unit of four (EURASIA_TAPDSSTATE_- + * USEATTRIBUTESIZE_ALIGNSHIFT 4 bytes-log2, sgxdefs.h:919-921): what the + * block's DMA carries, rounded up, as the vendor sizes it from its state + * dword count (opengles2/validate.c:2925, 3075). The draw record's word 1 + * carries it in [7:0]; the capture's four covered fifteen. */ +unsigned xpsb_heap_state_regs(const uint32_t *h); +/* The DOUTD control word that carries a block of n dwords into the primary + * attributes. A burst is at most sixteen dwords and a transfer at most + * sixteen bursts (EURASIA_PDS_DOUTD1_BSIZE_MAX, BLINES_MAX, sgxdefs.h:4226- + * 4235), so a count without a factor of sixteen or less - 17, 19, 23 - is + * carried as the next count that has one, and *carried says how many that + * is. The vendor's EncodeDmaBurst() (opengles2/validate.c:255-300) splits + * such a count into a second transfer instead; the state descriptor here + * has room for one, and the extra dword lands in an attribute the copy + * program never reads. */ +uint32_t xpsb_state_dma_ctl(unsigned n, unsigned *carried); +/* Heap dword of a group's first word, or -1 when the mask does not name it. */ +int xpsb_heap_state_off(const uint32_t *h, unsigned group); +/* Put a group in (on, from words[]) or take it out, moving every later group, + * the descriptor's count and, when the room runs out, the block itself. A + * group already present is overwritten in place. Returns -ENOSPC when it + * cannot fit and -EINVAL for a group the block does not know; the heap is + * untouched on either. */ +int xpsb_heap_state_set(uint32_t *h, unsigned group, int on, + const uint32_t *words); + +/* ISP word A (EURASIA_ISPA_*, sgxdefs.h:1360-1381). Bits 24:22 are the depth + * compare function and bit 20 the depth write-disable; bit 25 enables + * blending. The three bits below say which further words follow it in the + * block, and xpsb_heap_set_isp() lays them out from them. */ +#define XPSB_ISP_BLEND (1u << 25) +#define XPSB_ISP_2SIDED (1u << 11) /* EURASIA_ISPA_2SIDED, :1375 */ +#define XPSB_ISP_BPRES (1u << 9) /* EURASIA_ISPA_BPRES, :1377 */ +#define XPSB_ISP_CPRES (1u << 8) /* EURASIA_ISPA_CPRES, :1378 */ + +/* The USE program the vertex data master runs to hand the block to the MTE: + * "mov o0, pa0" repeated for the block's dwords - a second one from o16 past + * sixteen, the repeat's limit (EURASIA_USE_MAXIMUM_REPEAT) - then "emitst + * #n" naming the count, as the vendor's USEGenWriteStateEmitProgram() writes + * it (codegen/usegen/usegen.c:109-150, 592-700). The count is exact: the + * captured program copies and emits the captured fifteen, and a block one + * dword longer was handed over one dword short. One program per count lives + * in the USSE image from XPSB_USSE_STATE_OFF, a slot each; the frame's and + * each record's DOUTU name the one their block needs. */ +#define XPSB_USSE_STATE_OFF 0x2060 +#define XPSB_USSE_STATE_SLOT 0x20 +#define XPSB_USSE_STATE_MAX XPSB_STATE_MAX_DWORDS +#define XPSB_USSE_STATE_END (XPSB_USSE_STATE_OFF + \ + XPSB_USSE_STATE_MAX * XPSB_USSE_STATE_SLOT) +/* Write the program for n dwords at slot; returns its size in bytes, 0 for + * a count the block cannot have. */ +unsigned xpsb_usse_state_copy(uint32_t *slot, unsigned n); +/* Where the program for n dwords is in the USSE image, and its size. */ +unsigned xpsb_usse_state_copy_off(unsigned n); +unsigned xpsb_usse_state_copy_size(unsigned n); + +/* Install one ISP state: ff[0..2] are words A, B and C of the front set, bf + * the back set's, read only under 2SIDED. B goes in under BPRES and C under + * CPRES; a back set carries all three, which is what the vendor emits + * (opengles2/validate.c:2644-2677). Groups the words do not call for are + * taken out. */ +int xpsb_heap_set_isp(uint32_t *h, const uint32_t *ff, const uint32_t *bf); + +/* The PDS state pointer group: the secondary program's binding, the pixel + * data-master control word and the primary program's binding. */ +#define XPSB_PDS_W_SEC 0 +#define XPSB_PDS_W_CTL 1 +#define XPSB_PDS_W_PRI 2 +unsigned xpsb_heap_pds_dw(const uint32_t *h, unsigned w); + +/* Group 10 is the MTE's output selects. The vertex size is + * EURASIA_MTE_VTXSIZE, whose clear mask on this core is 0x80FFFFFF - seven + * bits at 24, not the five this once masked. Five truncated any record of + * thirty-two floats or more, which nothing has reached yet but which the + * field plainly allows. */ +#define XPSB_STATE_G10_DW 0x7f000000u /* record dwords in [30:24] */ +#define XPSB_STATE_G10_SHIFT 24 +/* Bit 8 of the same word is the DDK's EURASIA_MTE_SIZE: the record + * carries a point size for the MTE to read, beside WPRESENT at 12, the + * base colour at 11, the offset colour at 10 and the fog at 9. Only the + * width field above was ever written here, so every presence bit in this + * word is still whatever the captured frame carried. */ +#define XPSB_STATE_G10_SIZE (1u << 8) +#define XPSB_STATE_G10_OFFSET (1u << 10) /* EURASIA_MTE_OFFSET */ + +/* USSE programs live in 32-byte slots in buffer 2. Slot 0x0100 is the + * fragment program - one instruction in the captured frame, four slots' worth + * of room for the composite shader's two. */ +#define XPSB_USSE_FRAG_OFF 0x0100 +#define XPSB_USSE_SLOT_DW 8 +/* What xpsb_composite_shader() builds: two instructions. */ +#define XPSB_COMPOSITE_DW 4 +#define XPSB_USSE_DW (0x2360 / 4) /* the slots, then the state copies */ +#define XPSB_USSE_NSLOT 16 +#define XPSB_FRAG_DEFAULT_LO 0xa0000080u +#define XPSB_FRAG_DEFAULT_HI 0x81840005u +#define XPSB_CLEAR_DEFAULT 0x1a1a26u + +/* Slot for the precompiled YUV program of one FOURCC, past the sixteen the + * captured frame carries. 0x80 bytes holds the longest of them, packed YUV's + * fourteen instructions. */ +#define XPSB_USSE_VID_OFF 0x0200 +#define XPSB_USSE_VID_SLOT (0x80 / 4) + +/* Source register field of a PCKUNPCK instruction, w0[13:7]. The packed YUV + * program reads its texel from pa0, which is the texture sample only under the + * closed module's iterator-free plumbing; this frame layer iterates a colour + * into pa0 and the sample lands at pa1. Route B of + * disasm/video-integration.md section 1. */ +#define XPSB_PCK_SRC1_MASK 0x00003f80u +#define XPSB_PCK_SRC1_SHIFT 7 +#define XPSB_VID_SRC_PA 1 + +/* ds0[1] of the primary PDS program: the USE task's temporary-register + * allocation, at [31:27], with the rest of the count in XPSB_HEAP_USE_TEMPS_HI. + * + * [corrected 2026-08] An earlier note here read this field as gating access to + * the register file - "at 1 << 27 a three-register program sees r2 alias r0". + * That measurement is retracted: re-run with a known-good frame before each + * reading, r0..r31 are writable at ds0[1] = 0 exactly as at 4, and the original + * numbers were each reading what the previous measurement left behind. See + * work/usse-hwexec/README.md, "A claim that did not survive checking". + * What the pair does do is carry a count past 31: a fragment program needing + * exactly 32 - low five bits zero, so the whole count is in the high word - + * renders correctly, which it cannot do if the high word is ignored. */ +#define XPSB_HEAP_USE_TEMPS 0xd1 +#define XPSB_HEAP_USE_TEMPS_HI 0xd8 +#define XPSB_VID_TEMPS 0x30000000u + +/* The eleven conversion floats, in the free 64-byte block after the secondary + * PDS program: heap dwords 0x110..0x11a. heap_init[]'s first block stops at + * 0xfb, XPSB_SEC_PDS_OFF occupies 0x100..0x10d, and no relocation names any + * heap offset between 0x400 and 0x43e4. */ +#define XPSB_VID_CONST_OFF 0x440 +#define XPSB_VID_CONST_DW (XPSB_VID_CONST_OFF / 4) +#define XPSB_VID_CONST_N XPSB_VIDSHADER_SA_PACKED + +/* Secondary attributes. The pixel PDS binding at heap dword 0x1149 names a + * secondary PDS program, whose job is to load the USE secondary attribute + * bank; the captured stream points it at heap+0x380 with a data size of 0, + * which is the bare HALT sitting there. A program that carries constants needs + * a data segment, and the code segment starts at data + 4 * size - which from + * 0x380 would land on the vertex-fetch descriptor at heap+0x3a0. So a loading + * program moves to XPSB_SEC_PDS_OFF, the first 64-byte block past the heap's + * index buffer and named by nothing else in the heap or the relocations. + * + * A second sampled unit grows the primary program's data segment to 16 dwords, + * so its code segment starts at heap+0x380 as well and the bare HALT moves to + * the same free block. The capture did the same thing, relocating the + * secondary program from 0x20010380 to 0x200103a0 between P_mt2 and P_mt3. + * + * Binding word 0 carries the program address in bits 23:0 and its data size in + * dwords in bits 31:24; word 1 carries the secondary attribute allocation in + * bits 24:18, in blocks of 128 - XpsbFlushParamblock's (n + 0x7f) >> 7 - and + * in bits 3:0 the number of iterated varyings besides position, which is the + * iterated colour plus one texture coordinate set per sampled unit. Word 2 is + * the primary program: address in bits 23:0, data size in dwords in 31:24. */ +#define XPSB_SEC_PDS_NULL 0x380 +#define XPSB_SEC_PDS_OFF 0x400 +#define XPSB_SEC_PDS_MAX XPSB_VIDSHADER_SA_PDS_MAX +#define XPSB_SA_BLOCK_MASK 0x01fc0000u +/* The pixel data-master control word's fields, as the DDK names them: + * EURASIA_PDS_PIXELSIZE [6:0], SECATTRSIZE [24:18], USETASKSIZE [17:16] and + * PDSTASKSIZE [29:25]. The pixel size is the primary attribute registers one + * pixel costs, unscaled on this core (its align shift is zero). */ +#define XPSB_VARYING_MASK 0x0000007fu +#define XPSB_DMS_PIXELSIZE_MASK 0x0000007fu +#define XPSB_DMS_USETASK_MASK 0x00030000u +#define XPSB_DMS_USETASK_SHIFT 16 +#define XPSB_DMS_PDSTASK_MASK 0x3e000000u +#define XPSB_DMS_PDSTASK_SHIFT 25 + +/* Build that word for a shader costing pixel_regs primary attribute registers + * and temps temporaries, with sa_dwords of secondary attributes. */ +uint32_t xpsb_pds_dms_word(uint32_t cur, unsigned pixel_regs, unsigned temps, + unsigned sa_dwords); +#define XPSB_PDS_HALT 0xaf000000u + +/* The primary (texture-sampling) PDS program: data segment at its own base, + * code at data + 4 * size. The texture state words above are its ds0[2 + 2u] + * and ds0[3 + 2u], and the address slots the relocations target its + * ds1[2 + 2u]. + * + * The capture put it at 0x340, where the vertex-fetch descriptor at 0x3a0 + * left it 96 bytes for data and code together - three sampled units, and a + * fourth issue's address slot lands on memory dword 24 of the sixteen that + * fits. It lives instead in the block above the video constants, which no + * relocation and nothing else in the heap names (see the secondary-attribute + * note above), so the list is bounded by the program and not by where the + * capture happened to put it. RC_PRI carries the address. */ +#define XPSB_PRI_PDS_OFF 0x340 +#define XPSB_PRI_PDS_END 0x3a0 +#define XPSB_PRI_PDS_DW (XPSB_PRI_PDS_OFF / 4) +/* Where a list that does not fit the captured slot goes instead, and how far + * it may run. A frame that fits stays byte-identical to the capture. */ +#define XPSB_PRI_PDS_ALT 0x480 +/* 0x100 is what the widest list the arrays hold needs: eight issues reach + * memory dword 40, and their code is eighteen more - 236 bytes. Sized to that + * rather than to the whole free block, because the per-record copies step by + * it and a wider step runs them into the vertex PDS copies above. */ +#define XPSB_PRI_PDS_ALT_END 0x580 + +/* Unit u's state and address dwords, from the base the frame's program is + * built at. The data segment's banks alternate every eight dwords, so these + * are not a fixed stride apart once u reaches three - stepping them by two + * put unit 3's words on top of unit 1's. */ +/* Where unit u's state words are staged. The captured slot has room for three + * units before ds0[8] lands on the bare HALT at XPSB_SEC_PDS_NULL and the + * vertex descriptor behind it - writing unit 3 there left the secondary + * program running a texture control word, which stalls the core with no MMU + * fault. A unit past the third only exists on a program too wide for that slot + * anyway, so its words go where that program does. */ +static inline unsigned xpsb_tex_unit_base(unsigned u) +{ + return u < 3u ? XPSB_PRI_PDS_OFF : XPSB_PRI_PDS_ALT; +} + +static inline unsigned xpsb_texctl_dw_at(unsigned base, unsigned u) +{ + return base / 4 + xpsb_pds_ctl_dw(u); +} +static inline unsigned xpsb_texstate_dw_at(unsigned base, unsigned u) +{ + return base / 4 + xpsb_pds_fmt_dw(u); +} +static inline unsigned xpsb_texaddr_dw_at(unsigned base, unsigned u) +{ + return base / 4 + xpsb_pds_addr_dw(u); +} + +/* The vertex-fetch descriptor, the first thing in use past the primary PDS + * program's own 64-byte block (see the secondary-attribute note above). It is + * what bounds how far the primary program may grow once the bare HALT at + * XPSB_SEC_PDS_NULL has been moved out of the way. */ +#define XPSB_PDS_VTXDESC_OFF 0x3a0 + +/* The vertex USE program emits the record the vertex DMA fetched, and its + * dword count less one sits in bits [47:44] of the first instruction: all + * seven DMA-and-program pairs the captured heap carries agree. */ +#define XPSB_USSE_VTX_OFF 0x0140 +#define XPSB_USSE_VTX_MASK 0x0000f000u + +/* Where the present blit puts the frame on the panel, from the capture. */ +#define XPSB_PRESENT_X 60 +#define XPSB_PRESENT_Y 80 + +/* Hardware surface formats, from disasm/emit-pixel-shader.md section 3.1. The + * same five-bit code selects the render target and the sampled texture. */ +#define XPSB_FMT_A8 0x00 +#define XPSB_FMT_4444 0x02 +/* Two interleaved 8-bit channels - gl-re/textures.md's al88. */ +#define XPSB_FMT_AL88 0x07 +#define XPSB_FMT_1555 0x04 +#define XPSB_FMT_565 0x05 +/* The 16-bit and float texel codes, EURASIA_PDS_DOUTT1_TEXFORMAT_* + * (sgxdefs.h:4586-4598), one or two channels a plane; sampled only. */ +#define XPSB_FMT_U16 0x09 +#define XPSB_FMT_S16 0x0a +#define XPSB_FMT_F16 0x0b +#define XPSB_FMT_8888 0x0c +#define XPSB_FMT_BGR8888 0x0d +#define XPSB_FMT_U1616 0x0f +#define XPSB_FMT_S1616 0x10 +#define XPSB_FMT_F1616 0x11 +#define XPSB_FMT_F32 0x12 +/* ETC1: EURASIA_PDS_DOUTT1_TEXFORMAT_PVRTIII (sgxdefs.h:4639), eight bytes a + * 4x4 block in the twiddled layout only, so it has no bytes-per-pixel. */ +#define XPSB_FMT_ETC1 0x1b +#define XPSB_FMT_YUY2 0x1c +#define XPSB_FMT_UYVY 0x1d + +/* Top byte of the generic surface descriptors at heap 0x04b/0x10033/0x34c: + * bits 31:29 are 0b011, non-mipmapped and untwiddled. */ +#define XPSB_SURF_FMT(f) (0x60000000u | ((uint32_t)(f) << 24)) +#define XPSB_SURF_FMT_8888 XPSB_SURF_FMT(XPSB_FMT_8888) + +/* Sampler modes, in XpsbAddrModes / XpsbFilterFormats order. */ +#define XPSB_WRAP_REPEAT 0 +#define XPSB_WRAP_CLAMP 1 +#define XPSB_WRAP_CLAMPGL 2 +#define XPSB_WRAP_MIRROR 3 +/* The three border forms, EURASIA_PDS_DOUTT0_ADDRMODE_CLAMPBDR (5), + * CLAMPBDRMEM (6) and REPEATBDRMEM (4), sgxdefs.h:4411-4413. The two MEM + * forms read a border map laid out in front of the texture - + * xpsb_border_map_texels() - and are refused without one; CLAMPBDR carries no + * MEM suffix, reads no map, and what it clamps to is not stated anywhere. + * + * Nowhere in the DDK is any of the three selected: the constants above are + * the only lines in the whole tree that name them, and the ES1 and ES2 + * drivers translate every GL wrap into REPEAT, FLIP or CLAMP + * (opengles2/tex.c:1755-1815). So the colour CLAMPBDR supplies has no source + * in the DDK, and neither does which end of the map the unit is handed. It + * cannot come from state: DOUTT0 is allocated bit for bit, DOUTT2 is the + * address, and this core has three texture words and not the four later + * cores use (EURASIA_TAG_TEXTURE_STATE_SIZE, sgxdefs.h:5047-5053) - whose + * fourth word is a swizzle and a LOD adjust, not a colour. A hardware + * constant was what was left - and the part does not supply one: measured + * 2026-09-03, a frame with CLAMPBDR selected and no map comes back with + * neither the draw nor the clear in it, which is what code 7 does too. + * + * The displacement was then measured at 2x2, 4x4 and 8x8: it is the DDK's own + * xpsb_border_map_face_offset() for the texture's size, the last of the three + * to the byte. So the address in DOUTT2 is the map's base and the texels + * begin at that offset - and with them laid there the part renders the + * texture correctly and then clamps to the edge. No border texel is fetched, + * from the map or from anywhere else, and no field beside the address mode + * could enable one. The codes are an address, not a border; + * work/fix-border/README.md has the runs. */ +#define XPSB_WRAP_CLAMP_BORDER 5 +#define XPSB_WRAP_CLAMP_BORDER_MAP 6 +#define XPSB_WRAP_REPEAT_BORDER_MAP 7 +#define XPSB_FILTER_NEAREST 0 /* the unit's POINT, below */ + +/* Between levels, word 0 bit 9: nearest picks one level, linear blends two. + * Orthogonal to the within-level and magnification filters. */ +#define XPSB_MIPFILTER_NEAREST 0 +#define XPSB_MIPFILTER_LINEAR 1 + +/* hwtextype, word 1 bits 31:29 - the target class and the layout. */ +#define XPSB_SURF_2D_TWIDDLED 0x00000000u +#define XPSB_SURF_2D_LINEAR 0x60000000u + +/* Sampler word 0 base. The two drivers disagree on bits 25:21 and bit 1; see + * xpsb_tex_state_mode() for which one is the default and why. */ +#define XPSB_TEXCTL_CAPTURE 0 +#define XPSB_TEXCTL_XPSB 1 + +/* Texture types, DOUTT1[31:29] - the DDK's EURASIA_PDS_DOUTT1_TEXTYPE_*. + * 2D, 3D and CEM are log2-size (twiddled) encodings; STRIDE is the linear one + * the non-twiddled branch builds. 3D is sgxdefs.h:4557, inside + * SGX_FEATURE_VOLUME_TEXTURES, which the SGX535 block at + * sgxfeaturedefs.h:193 defines; its depth is SSIZE, DOUTT1[15:12]. */ +#define XPSB_TEXTYPE_2D 0u +#define XPSB_TEXTYPE_3D 1u +#define XPSB_TEXTYPE_CEM 2u +#define XPSB_TEXTYPE_STRIDE 3u + +/* Submission parameters, all from the captured frame. Sizes are dword counts, + * not bytes - psb_submit_copy_cmdbuf() computes cmd_offset + (cmd_size << 2). */ +#define XPSB_SCENE_NB 4 +#define XPSB_TA_ENGINE 3 +#define XPSB_TA_TAFLAGS 3 +#define XPSB_TA_CMD_OFF 56 +#define XPSB_TA_TA_OFF 0 +#define XPSB_TA_OOM_OFF 264 +#define XPSB_TA_RELOC_OFF 0 +#define XPSB_RAS_ENGINE 2 +#define XPSB_RAS_TAFLAGS 3 +#define XPSB_RAS_CMD_OFF 131072 +#define XPSB_RAS_TA_OFF 131072 +#define XPSB_RAS_TA_SIZE 0 +#define XPSB_RAS_OOM_OFF 131072 +#define XPSB_RAS_OOM_SIZE 0 +#define XPSB_RAS_RELOC_OFF 0x8000 + +#define XPSB_RELOC_OP_OFFSET 0 +#define XPSB_RELOC_OP_USE_OFFSET 4 +#define XPSB_RELOC_OP_USE_REG 5 + +#define XPSB_NUM_TA_RELOCS 77 +/* one more record per address slot past the first, and one for the secondary + * PDS program's constant block. Sized by the issue list rather than by the two + * units the capture sampled: a slot left unrelocated keeps the placeholder + * address, which nothing is bound at, and the render faults on it. */ +#define XPSB_MAX_TA_RELOCS (XPSB_NUM_TA_RELOCS + XPSB_MAX_TEX) +#define XPSB_NUM_RAS_RELOCS 16 +#define XPSB_RAS_LIST_LEN 6 + +/* The level-of-detail adjust the TAG adds to the computed LOD: DOUTT0[26:21], + * signed 3.3 fixed point biased by 31, so -3.875 to +4.0 in eighths. The name + * and the encoding are the DDK's EURASIA_PDS_DOUTT0_DADJUST, whose _ZERO_UINT + * is the 31 below; the 8-bit form the defines also carry belongs to cores with + * SGX_FEATURE_8BIT_DADJUST, which this one is not. */ +#define XPSB_LOD_BIAS_ZERO 31u +#define XPSB_LOD_BIAS_MIN (-3.875f) +#define XPSB_LOD_BIAS_MAX (4.0f) +uint32_t xpsb_lod_bias_code(float bias); + +/* The texture unit's filter encodings. ANISO is not a ratio - the ratio is + * its own field - but the filter mode that makes the unit read that ratio. */ +#define XPSB_FILTER_POINT 0u +#define XPSB_FILTER_LINEAR 1u +#define XPSB_FILTER_ANISO 2u +#define XPSB_FILTER_ANISOPOINT 3u + +/* Mirrors XpsbSurface: stride is in BYTES, format is one of XPSB_FMT_*, and + * the sampler fields are only read for a texture. */ +struct xpsb_surface_desc { + uint32_t handle, offset, w, h, stride, format; + uint32_t umode, vmode, minfilter, magfilter; + /* Mipmapping. nlevels 0 and 1 both mean a single level, so a + * zero-initialised descriptor keeps the non-mipmapped encoding. + * A twiddled surface has no pitch and must be power of two. */ + uint32_t nlevels, mipfilter, twiddled; + /* The LOD adjust, already encoded by xpsb_lod_bias_code(). Zero is + * left as the field's own zero so a zero-initialised descriptor keeps + * the captured word. */ + uint32_t lod_bias; + /* The anisotropic ratio, DOUTT0[16:14]: 0 none, then 2x, 4x, 8x, 16x. + * The ratio alone does nothing - the min and mag filter fields have to + * name an anisotropic mode as well, which this does for them. */ + uint32_t aniso; + /* The texture type, DOUTT1[31:29], as the DDK's + * EURASIA_PDS_DOUTT1_TEXTYPE_*. Zero keeps what the twiddled and + * linear branches below already encode, so an ordinary descriptor + * need not set it. */ + uint32_t textype; + /* The highest level the TAG may sample. Only read when max_level_set + * is non-zero, because zero is a legal clamp - it is the vendor's own + * MIPMAPCLAMP_MIN - and could not otherwise be told from "unset". */ + uint32_t max_level, max_level_set; + /* A volume: depth slices behind each level, log2 in DOUTT1 SSIZE + * [15:12] (sgxdefs.h:4730-4744) with the texture type 3D, and the + * third axis wrapped by DOUTT0 SADDRMODE [2:0] (sgxdefs.h:4440-4449). + * 0 and 1 are a 2D texture, whose [2:0] stay at the captured word's. + * smode is an XPSB_WRAP_* index like umode. Twiddled only: every + * volume encoding is a log2 one. */ + uint32_t depth, smode; + /* DOUTT0 bit 27, EURASIA_PDS_DOUTT0_GAMMA (sgxdefs.h:4344): the unit + * converts the texel from sRGB before it filters. SGX535 has the + * feature (SGX_FEATURE_GAMMACORRECT_TEXTURES, sgxfeaturedefs.h:192), + * whose own description is "Gamma correction is supported on texture + * reads" (:44-46). One bit here; on 543 the same position is a + * two-value field, GAMMA_R and GAMMA_GR (sgxdefs.h:4322-4326), which + * names channels from R upwards and never alpha. No vendor code path + * sets the bit on this core. */ + uint32_t gamma; + /* DOUTT0 bit 30, CHANREPLICATE (sgxdefs.h:4335), on a format the + * one-byte rule below does not already set it for. The vendor sets it + * for A8L8 too (sgx535pixfmts.h:603, opengles2/texmgmt.c:3466). */ + uint32_t chanrep; + /* A border map is laid out in front of the texels: offset names the + * map and the texels are xpsb_border_map_face_offset(w) texels further + * on - the offset the unit itself adds, measured 2026-09-03 at three + * sizes (EURASIA_TAG_BORDERMAP_*, sgxdefs.h:5061-5077). Needed by the + * two *_BORDER_MAP address modes, refused without. */ + uint32_t border_map; +}; + +/* Where the map's face for a square power-of-two texture of that top-level + * size starts, in texels: EURASIA_TAG_BORDERMAP_OFFSET_x on this core + * (sgxdefs.h:5064-5075), which is 48 + 8 * size in closed form. Zero for a + * size the table has no entry for - anything above EURASIA_TEXTURESIZE_MAX, + * which is 2048 here (sgxdefs.h:423), and anything not a power of two. */ +uint32_t xpsb_border_map_face_offset(uint32_t size); + +/* The table's whole extent for that size, 48 + 16 * size texels: its own face + * is 8 * size and sits at the offset above. Not the layout - the unit fetches + * at the face offset, measured - and kept for the prefix-sum checks. */ +uint32_t xpsb_border_map_texels(uint32_t size); + +/* The 3D twiddle. The DDK names the type and the size field and nothing + * about the order of texels in a volume; the two orders below are the + * candidates, and which the unit walks is settled on hardware + * (work/feat-texunit/TESTPLAN.md). MORTON interleaves the three axes a bit + * at a time, y then x then z, exactly as the 2D order interleaves y and x, + * with an axis that runs out of bits dropping out of the rotation and the + * last one's excess linear on top - the 2D rule with a third axis. SLICES is + * the 2D twiddle per slice, slices consecutive. */ +enum xpsb_twiddle3 { + XPSB_TWIDDLE3_MORTON = 0, + XPSB_TWIDDLE3_SLICES +}; +uint32_t xpsb_twiddle3_index(uint32_t x, uint32_t y, uint32_t z, + uint32_t log2w, uint32_t log2h, uint32_t log2d, + enum xpsb_twiddle3 order); +/* Texels from the start of a chain to the given level of a volume: each + * level is w*h*d texels and every axis halves to a floor of one. */ +uint32_t xpsb_twiddle3_level_offset(uint32_t w, uint32_t h, uint32_t d, + uint32_t level); +/* A whole level between its twiddled form and a linear volume of + * src_stride bytes per row and src_slice bytes per slice. */ +int xpsb_twiddle3_level(void *dst, const void *src, uint32_t w, uint32_t h, + uint32_t d, uint32_t src_stride, uint32_t src_slice, + unsigned bpp, enum xpsb_twiddle3 order); +int xpsb_untwiddle3_level(void *dst, const void *src, uint32_t w, uint32_t h, + uint32_t d, uint32_t dst_stride, uint32_t dst_slice, + unsigned bpp, enum xpsb_twiddle3 order); + +struct xpsb_reloc { + uint32_t reloc_op, where, buffer, mask, shift; + uint32_t pre_add, background, dst_buffer, arg0, arg1; +}; + +struct xpsb_fb_desc { + uint64_t size, create, vflags, vmask; +}; + +struct xpsb_scene_arg { + int handle_valid; + uint32_t handle, w, h, num_buffers; +}; + +struct xpsb_bo { + uint32_t handle; + uint64_t size, create, gpu, token, start; + void *base, *cpu; +}; + +/* One draw block. The three heap byte offsets are both the placeholder words + * written into the draw record and the pre_add of its three relocations, so + * record and relocation cannot drift apart. */ +struct xpsb_draw_desc { + uint32_t hdr_bg, state, seq, idx, prim, kind, nidx; +}; + +/* What the relocation emitter has to vary. first_draw drops that many leading + * draw blocks - 0 is the captured frame, 2 is the quad without the clear. */ +struct xpsb_reloc_cfg { + unsigned first_draw, ntex; + uint32_t dest_offset; + uint32_t tex_offset[XPSB_MAX_TEX]; + uint32_t sec_pds_off; /* 0 keeps the captured bare HALT */ + unsigned sec_pds_dwords; + /* Where the state block is: the byte offset its descriptor names and + * the dword its PDS pointer group starts at, both from the heap as + * laid out (xpsb_heap_state_base(), xpsb_heap_pds_dw()). Zero keeps + * the captured positions. */ + uint32_t state_off; + unsigned pds_dw; + /* The state-copy program the block's size calls for: its byte offset + * in the USSE image and its size, for the DOUTU's three relocations. + * Zero keeps the captured fifteen-dword program. */ + uint32_t state_use_off; + unsigned state_use_size; + /* Heap offset the secondary program's DOUTD reads its constants from, + * 0 when it carries none. Like every other heap-internal address in the + * heap it is a placeholder the kernel relocates. */ + uint32_t sec_const_off; + /* Fragment USE program: 0 keeps the captured slot at 0x100. size is the + * byte count psb_grab_use_base() reserves. */ + uint32_t frag_use_off; + unsigned frag_use_size; + /* Point the ISP's background object at the load-back record instead of + * the clearing one, so a tile the tiler did not bin comes back holding + * what the colour buffer already had rather than the clear. It is the + * same swap the out-of-memory chunk does - gl-re/scene-management.md + * section 5 - and it is what lets a caller draw over an existing + * picture instead of replacing it. */ + unsigned bg_load; + /* Where each sampled unit's address slot is, and how many issues the + * primary program carries. Two units may share one coordinate set and + * an iterated varying takes an issue of its own, so neither follows + * the unit count. Zero keeps one issue per unit, which is what the + * captured frames have. */ + unsigned nissue; + unsigned char tex_issue[XPSB_MAX_TEX]; + /* How many of those slots carry a texture address. The generator binds + * the first ntex to the caller's units; the rest are issues no unit + * patches - the dummy a textureless program carries, or a third unit + * past the two the capture had - and they take unit 0's buffer so that + * they name mapped memory. Zero relocates ntex slots, as before. */ + unsigned ntexslot; + /* Where the primary program was built, when that is not the captured + * slot: the address RC_PRI names, and the base RC_FRAG's DOUTU words + * and RC_TEX's address slots are counted from. Zero is the capture. */ + uint32_t pri_pds_off; + /* Dword count of the primary program the issue list actually built. + * Zero sizes it from ntex instead, which is right only while the plain + * generator is the one that built it. */ + unsigned pri_pds_dwords; +}; + +/* Screen-space rectangle with its texture rectangle, drmquad.c:748-758's four + * vertices reduced to the tuple the composite entry points hand over. */ +struct xpsb_quad { + float x0, y0, x1, y1; + float u0, v0, u1, v1; + float r, g, b, a; + float m0, n0, m1, n1; /* mask coordinates, texture unit 1 */ +}; + +struct xpsb_frame { + int fd; + int w, h; + uint32_t stride; /* destination stride in pixels */ + int depth_test; + int texctl_mode; + int blend, blend_op; + int scalar_on; + uint32_t scalar; + int video_on; + uint32_t video_fourcc; + float video_conv[XPSB_VID_CONST_N]; + struct xpsb_shader_flags blend_flags; + unsigned nta, ncmd, noom, nras; /* stream lengths in dwords */ + unsigned nta_reloc, nras_reloc; + unsigned ndraw, draw_cmd_off, nidx; + struct xpsb_bo bo[XPSB_FRAME_NBUF + 1]; + struct xpsb_bo relbo; + uint32_t vhandle[XPSB_FRAME_NBUF + XPSB_NTEX_MAX - 1]; + struct xpsb_surface_desc dest, tex[XPSB_MAX_TEX]; + struct xpsb_reloc_cfg rcfg; + struct xpsb_scene_arg scene; +}; + +int xpsb_frame_init(struct xpsb_frame *f, int drmfd, int w, int h); +void xpsb_frame_fini(struct xpsb_frame *f); +int xpsb_frame_submit(struct xpsb_frame *f); + +/* Caller-supplied surfaces. Both refuse, with EINVAL, a surface whose stride + * is not bpp * ALIGN(w,32) - that refusal is what makes the caller fall back + * to the CPU path, so it must never be ignored. Neither rebuilds the frame's + * geometry-dependent streams; call xpsb_frame_set_geometry() first if the + * destination changes size. */ +int xpsb_frame_set_dest(struct xpsb_frame *f, const struct xpsb_surface_desc *d); +int xpsb_frame_set_texture(struct xpsb_frame *f, int unit, + const struct xpsb_surface_desc *t); + +/* Which sampler word 0 the setters build. XPSB_TEXCTL_CAPTURE, the default, + * is the word the working stream carries; XPSB_TEXCTL_XPSB is the blob's. */ +int xpsb_frame_set_texctl_mode(struct xpsb_frame *f, int mode); + +/* Number of sampled texture units. Only 1 can be submitted - see the comment + * on xpsb_frame_set_ntex() in xpsb_frame.c - so 2 exists to build and check + * the tables, and xpsb_frame_submit() refuses while it is selected. */ +int xpsb_frame_set_ntex(struct xpsb_frame *f, unsigned n); + +/* One quad into the user draw record. */ +int xpsb_frame_set_quad(struct xpsb_frame *f, const struct xpsb_quad *q); + +/* Blending. Absent from the captured stream, which leaves the ISP word at + * 0x01d00000 with enable bit 25 clear and a one-instruction fragment program. + * set_composite sets the bit and writes xpsb_composite_shader()'s two SOP2 + * words into the fragment slot; set_blend(f, 0) restores both. Unblended is + * the default. */ +int xpsb_frame_set_composite(struct xpsb_frame *f, int op, + const struct xpsb_shader_flags *fl); +int xpsb_frame_set_blend(struct xpsb_frame *f, int on); + +/* The constant operand of a scalar source or a scalar mask: the caller's + * ARGB8888 pixel, verbatim, in sa0. It is the composite shader's other operand + * whenever xpsb_composite_shader() emits one of the sa0 modulate words, so it + * has to be set for exactly those and cleared for the rest; off restores the + * captured no-constant binding. */ +int xpsb_frame_set_scalar(struct xpsb_frame *f, int on, uint32_t argb); + +/* The packed YUV path: the FOURCC's precompiled program in place of the + * fragment shader, its temporary-register count, and psbSetupConversionData's + * eleven floats in sa0..sa10. Only YUY2 and UYVY have a program this layer can + * submit; every other FOURCC fails with ENOSYS. fourcc 0 restores the captured + * fragment binding. Refuses while a scalar constant is bound, because both + * want the same secondary attribute bank. */ +int xpsb_frame_set_video(struct xpsb_frame *f, uint32_t fourcc, + const float *conv); + +/* Clearing draws. On (the default) is the captured frame; off drops draw + * records 0 and 1, the two clearing rastgeom records and their relocations, + * so the destination's prior contents survive outside the quad. Never + * exercised on hardware - see the note in xpsb_frame.c. */ +int xpsb_frame_set_clear(struct xpsb_frame *f, int on); + +/* Regenerate the size-dependent buffers. stride is in pixels, 0 selects the + * round_up(w, 32) the vendor driver uses; depth_test enables the hardware + * depth compare in the per-draw ISP word. */ +int xpsb_frame_set_geometry(struct xpsb_frame *f, int w, int h, + uint32_t stride, int depth_test); + +/* CPU access to a buffer object, bracketed by the BO_MAP/BO_UNMAP ioctls + * separately from the mmap. bo[idx].cpu is valid between the two. */ +int xpsb_frame_map(struct xpsb_frame *f, unsigned idx); +void xpsb_frame_unmap(struct xpsb_frame *f, unsigned idx); + +/* Copy the rendered surface out row by row at the render-target stride. + * dst_stride is in bytes, 0 means w * 4. */ +int xpsb_frame_readback(struct xpsb_frame *f, void *dst, uint32_t dst_stride); + +/* Generators. Pure functions of the geometry, so they can be checked against + * drmcube's originals without a device. Counts are returned in dwords. */ +void xpsb_gen_heap(uint32_t *h, int w, int hgt, uint32_t stride, int depth_test); +void xpsb_gen_rastgeom(uint32_t *b, int w, int h); +/* Collapse the two clearing records to zero area. The frame keeps its shape - + * the state blocks the records carry are still walked, which the user draw + * turns out to need - but they cover no pixel, so what is already in the + * target survives. It is what a caller drawing over an existing picture + * needs, and dropping the records instead leaves the tiler unfinished. */ +void xpsb_rastgeom_clear_none(uint32_t *b); +/* The three records decoded, from the mapped buffer at submit: positions in + * pixels, the (u,v) they carry, and the header words. What reached memory, + * not what the generator meant to write. */ +void xpsb_rastgeom_dump(FILE *f, const uint32_t *b); +void xpsb_gen_draw_records(uint32_t *b); +unsigned xpsb_gen_ta_stream(uint32_t *p); +unsigned xpsb_gen_ta_raster_stream(uint32_t *p, int w, int h); + +/* Multisampling. + * + * The core reaches one sample per pixel and 2x2, and nothing between. + * SGXAddRenderTarget() accepts 1x1, 2x1, 1x2 and 2x2 + * (sgxrender_targets.c:740-751), but the two two-sample forms are behind + * SGX_FEATURE_MSAA_2X_IN_X and SGX_FEATURE_MSAA_2X_IN_Y, and the SGX535 block + * of sgxfeaturedefs.h (lines 189-232) defines neither - only SGX543, SGX544, + * SGX545 and SGX554 do. So two samples a pixel is not reachable on this part + * and XPSB_MSAA_2X does not exist; four is. + * + * Sample positions are programmable, four of them, one nibble an axis in + * sixteenths of a pixel: EUR_CR_ISP_MULTISAMPLECTL (0x04c8) and its MTE twin + * EUR_CR_MTE_MULTISAMPLECTL (0x0250) carry X0 at [3:0], Y0 at [7:4], X1 at + * [11:8] and so on to Y3 at [31:28] (sgx535defs.h:874-898, 1441-1465). The + * two must agree: the MTE iterates at the positions the ISP tested. + * + * The values are the vendor's, not invented here. One sample under + * SGX_ADDRTFLAGS_USEOGLMODE is the pixel centre, (8,8) + * (sgxrender_targets.c:904-907) - which is what the captured stream already + * carries as 0x00000088. Four is the rotated grid (6,2) (14,6) (2,10) + * (10,14) (sgxrender_targets.c:1015-1030), the non-upscaled branch, since + * nothing here asks for SGX_UPSCALING. */ +#define XPSB_MSAA_1X 1u +#define XPSB_MSAA_4X 4u + +/* Samples along one axis: 1 for XPSB_MSAA_1X, 2 for XPSB_MSAA_4X. Zero for a + * count the part cannot reach, which is every other value. */ +unsigned xpsb_msaa_axis(unsigned samples); +/* The MULTISAMPLECTL word for a sample count, or 0 for an unreachable one. */ +uint32_t xpsb_msaa_positions(unsigned samples); + +/* Apply a sample count to the two streams. Both scan by register number, as + * the other setters do, so an edit to a stream cannot move a field silently. + * A count xpsb_msaa_axis() refuses leaves both streams untouched and returns + * -1: a frame must not be fired at a sample count the part will not tile. + * + * The TA half is EUR_CR_TE_AA (0x0204) bit 31 for two samples in x and bit 30 + * for two in y (sgx535defs.h:710-716, set in sgxkick_client.c:1067-1077), and + * the MTE's copy of the positions. + * + * The raster half is the ISP's copy of the positions, EUR_CR_3D_AA_MODE + * (0x042c) whose only field on this core is ENABLE + * (sgx535defs.h:1238-1241 - the VALUE_2X and VALUE_4X encodings other cores + * have do not exist here, and sgxrender_targets.c:988-991 falls back to the + * plain enable mask exactly for that), the ISP's render box and the ZLS + * extent. */ +int xpsb_ta_set_msaa(uint32_t *p, unsigned n, unsigned samples); +int xpsb_ta_raster_set_msaa(uint32_t *p, unsigned n, int w, int h, + unsigned samples); + +/* The largest render target the core will take, from EURASIA_RENDERSIZE_MAXX + * and MAXY. sgxdefs.h:390-424 gives 8192 for SGX545, 4096 for the SGX543 + * family and 2048 for everything else, which is the arm this part takes; + * SGXAddRenderTarget() refuses anything larger outright + * (sgxrender_targets.c:701-708). The scene cookie's twelve-bit extent fields + * would encode 4096, which is where the driver's old claim came from, but the + * ISP will not render it. */ +#define XPSB_RENDER_SIZE_MAX 2048u + +/* Tell the render pass whether its background object loads the tile back or + * clears it; goes with the RC_BG relocation. */ +void xpsb_ta_raster_set_bg_load(uint32_t *p, unsigned n, int on); + +/* The same question for depth: whether the pass loads the depth buffer back + * or starts every tile at the background depth. Rewrites register 0x0480, + * the ISP's Z load/store control - see the function's comment for the field + * map and how the load half was found. + * + * The enable is two-level: a master enable per plane, and a FORCE bit that + * applies it to every region whether or not the tiler marked its header. + * Either alone does nothing, which is what hid the load from a bit sweep. */ +#define XPSB_ZLS_LOADTILED 0x00000001u /* Z/S in memory is tiled */ +#define XPSB_ZLS_SLOADEN 0x00000002u /* master stencil load */ +#define XPSB_ZLS_ZLOADEN 0x00000004u /* master depth load */ +#define XPSB_ZLS_STORETILED 0x00000008u +#define XPSB_ZLS_ZSTOREEN 0x00000010u /* master depth store */ +#define XPSB_ZLS_FORCEZLOAD 0x00000020u /* load every region */ +#define XPSB_ZLS_FORCEZSTORE 0x00000040u /* store every region */ +#define XPSB_ZLS_SSTOREEN 0x00000080u /* master stencil store */ +#define XPSB_ZLS_LOAD_AND 0xffffffffu +#define XPSB_ZLS_LOAD_OR (XPSB_ZLS_ZLOADEN | XPSB_ZLS_SLOADEN | \ + XPSB_ZLS_FORCEZLOAD) +/* EUR_CR_ISP_BGOBJ bit 8: the background object's SPM mask plane, which a load + * pass must not use or it resets the depth the ZLS loaded. Bit 9 (ENABLEBGTAG) + * stays so the object still runs its colour load-back. */ +#define XPSB_BGOBJ_MASK 0x00000100u +void xpsb_ta_raster_set_z_load(uint32_t *p, unsigned n, int on); + +/* The values a frame begins from where nothing is loaded: the background + * object's depth (EUR_CR_ISP_BGOBJDEPTH, 0x04b8, a float) and its stencil + * (EUR_CR_ISP_BGOBJ 0x04bc bits [7:0]; bits 8 and 9 are left as the stream + * carries them). */ +void xpsb_ta_raster_set_bg(uint32_t *p, unsigned n, uint32_t depth, + unsigned stencil); +/* The depth the clearing records' vertices carry, as float bits: records + * first..1 are the ones that clear, and the present blit's is left alone. */ +void xpsb_rastgeom_set_z(uint32_t *b, unsigned first, uint32_t z); + +/* The pixel extent the ISP iterates is SGX_S_TA_PIXEL_EXTENT (0x0248), which + * the kernel programs from the scene cookie - it is not in this stream, so + * there is nothing here to patch. It follows the width and height the submit + * carries. */ + +/* Narrow the tiler and the ISP to a rectangle of tiles, first and last. Both + * ends are inclusive tile indices - registers 0x040c and 0x0410 are + * EUR_CR_ISP_RENDBOX1 and RENDBOX2 (sgx535defs.h:1180-1194), and the vendor + * writes the last as ceil(x1 / 16) - 1 (sgxtransfer_utils.c:3259-3273). */ +void xpsb_ta_raster_set_range(uint32_t *p, unsigned n, unsigned tx0, + unsigned ty0, unsigned tx1, unsigned ty1); +/* The last tile a dimension of dim pixels reaches, which is what RENDBOX2 + * takes. */ +#define XPSB_LAST_TILE(dim) ((((unsigned)(dim) + 15u) / 16u) - 1u) +unsigned xpsb_gen_oom_stream(uint32_t *p); +unsigned xpsb_gen_raster_stream(uint32_t *p, int w, int h); + +/* The 3D raster register list with the planar video route's ten coefficient + * pairs appended; sgx_coeffs NULL leaves it as it was. */ +unsigned xpsb_gen_ta_raster_stream_coeffs(uint32_t *p, int w, int h, + const uint32_t *sgx_coeffs); + +/* The sixteen USSE programs of the captured frame, clear being the 21-bit + * immediate of the clear-colour program. Returns the dword count written. */ +unsigned xpsb_gen_usse(uint32_t *u, uint32_t clear); + +/* Fragment slot, eight dwords. The composite form writes the shader's two + * instructions and zeroes the rest of the slot; the default form restores the + * captured single-instruction program. */ +int xpsb_usse_composite(uint32_t *slot, int op, + const struct xpsb_shader_flags *fl); +void xpsb_usse_default_frag(uint32_t *slot); + +/* The precompiled YUV program for a FOURCC into the video slot. Returns the + * dword count, 0 for a FOURCC xpsb_vidshader() has no program for. The words + * are the blob's, verbatim. */ +unsigned xpsb_usse_video(uint32_t *u, uint32_t fourcc); + +/* Move the packed program's three PCKUNPCK reads to primary attribute pa. + * Returns -1 for a FOURCC with no packed program, or a pa the field cannot + * hold. */ +int xpsb_usse_video_pa(uint32_t *u, uint32_t fourcc, unsigned pa); + +/* The two heap blocks the packed video path adds: the USE task's temporary + * register count and the eleven conversion floats. conv NULL clears both. */ +void xpsb_heap_set_video(uint32_t *h, const float *conv); + +/* Byte offset of the fragment USE program in the user draw's task control + * word, bits [18:8] as offset / 16. Three relocations overwrite the whole + * field, so this only keeps the placeholder and the relocation in step. */ +void xpsb_heap_set_frag_use(uint32_t *h, uint32_t off); + +/* The secondary PDS program that loads n constants into sa0 upwards, and the + * data-segment size the binding must declare for it. n == 1 puts the constant + * in the program itself; above that it is a DMA from const_addr. Returns the + * program length in dwords, 0 for an n the emitter cannot build. */ +unsigned xpsb_gen_sec_pds(uint32_t *out, const float *consts, unsigned n, + uint32_t const_addr, unsigned *dsize); + +/* Point the pixel PDS binding at a secondary program of dsize dwords at heap + * byte offset off, allocating nattr secondary attributes. All zero restores + * the captured binding. */ +void xpsb_heap_set_sec_pds(uint32_t *h, uint32_t off, unsigned dsize, + unsigned nattr); + +/* Blend enable, the only part of blending that lives in the heap. */ +void xpsb_heap_set_blend(uint32_t *h, int on); + +/* The extent the pixel back end is allowed to write, heap dword 0x003: the + * pixel-emit clip, XMAX in [11:0] and YMAX in [23:12], inclusive + * (EURASIA_PIXELBE1S1_XMAX_SHIFT 0, _YMAX_SHIFT 12, sgxdefs.h:7159-7164). The + * line stride is a separate word at 0x000 and always the surface's, so a frame + * built for a sub-rectangle names the rectangle here and leaves the stride + * alone - the vendor's own sub-rectangle render does exactly that + * (sgx_render_flip_test.c:1771-1785 against :4410-4412). Read it before + * narrowing it; put back what was read. */ +#define XPSB_HEAP_EXTENT 0x003u +uint32_t xpsb_heap_extent(const uint32_t *h); +void xpsb_heap_set_extent(uint32_t *h, unsigned w, unsigned hgt); +/* Put the cull/front-face/shade-model group into the per-draw state block, or + * take it back out. With on == 0 the block is left exactly as captured. */ +void xpsb_heap_set_cull(uint32_t *h, uint32_t word, int on); + +/* Vertex-record size in floats, and everything in the heap that has to agree + * with a sampled-unit count: the vertex DMA dword count and byte stride, the + * primary PDS program, and the two binding words that carry its data size and + * the iterated varying count. */ +unsigned xpsb_vtx_stride(unsigned ntex); +void xpsb_heap_set_ntex(uint32_t *h, unsigned ntex); + +/* Coordinate sets one draw may carry. Two, which is what the captures + * exercise: every frame that samples two units names sets 0 and 1, and both + * a texissue of 2 and a useissue of 2 were measured delivering nothing at + * all - ioquake3's lit surfaces rendered black either way round. A third + * varying does not need a third set; it comes over as the packed colour, + * which is how the captured two-unit frame carries its iterated colour while + * both sets go to the two units. */ +/* Three. The captured frame carries two because it binds two texture units, + * and a third varying was expected to ride as the packed colour - but that is + * eight bits a channel, so a float varying loses its value there. glmark2's + * ideas logo is the case: ideas-logo.frag reads vertex_normal, vertex_position + * and eye_direction, ten components across three sets, and with only two the + * third never arrived, its lighting saturated and the logo and its cast shadow + * rendered white instead of purple. The part addresses TC0..TC9. */ +/* Coordinate sets the part iterates. */ +#define XPSB_NCOORD_MAX 3 +/* Set entries a program may hold: the coordinate sets and one more + * carried on the V0 colour iterator, which is not a coordinate set and + * rides in the record's colour quad rather than widening it. */ +#define XPSB_NSET_MAX (XPSB_NCOORD_MAX + 1) + +/* What a draw hands the fragment stage. The record is position, colour, then + * each coordinate set in order at its own width; a set may be sampled by a + * texture unit, delivered to the program as floats, or both. + * + * The iterated colour always rides on the first issue, packed, which is the + * shape every capture has. */ +struct xpsb_attribs { + unsigned nset; + unsigned colour; /* the program reads the packed colour */ + struct { + uint8_t width; /* floats in the record, 2..4 */ + uint8_t sampled; /* texture units reading it */ + /* Texture units reading it that the record does not carry a + * coordinate for - a program that computes its own coordinate + * and samples from a register. They take no float out of the + * set, so `sampled` stays zero and the set keeps its full + * width, but each still needs an issue of its own: SMP reads + * the unit's texture state out of the primary attributes the + * DOUTT wrote, so a unit with no issue samples state nothing + * set up and stalls the core. */ + uint8_t units; + uint8_t iterated; /* the program reads it as floats */ + /* The program reads components of it past the coordinate, so + * a texture issue must not take a register out of the set. */ + uint8_t values; + /* The iterator divides by the set's third float before it + * delivers the coordinate. A projected sample cannot be done + * in the shader - SMP reads its coordinate from the iterator + * and never from the register file - so the divide has to + * happen on the way in. */ + uint8_t projected; + /* The third float is a volume's slice coordinate, not a + * divisor: the set goes to the unit as UVS + * (EURASIA_MTE_TEXDIM_UVS, sgxdefs.h:2064) rather than UVT. */ + uint8_t volume; + /* Deliver the set to the USE as half floats. The record is + * unaffected - the MTE reads a coordinate set as F32 and has + * no format field for it, only the component count of group + * 14 - so this is the iterator's delivery format alone, and + * what it halves is the primary attribute space and the + * per-pixel iteration into it: four components arrive in two + * registers instead of four. The program then reads two + * components per register at byte offsets 0 and 2. */ + uint8_t f16; + /* Carried by a colour iterator rather than a coordinate set: + * 1 for V0, 2 for V1. The group-14 rule reaches only the + * coordinate sets - a frame renders nothing once more than one + * set precedes the last and it is wider than a UV pair - and + * V0/V1 are not named there, so a set moved onto one is + * capacity that rule does not cost. Its floats then live in + * the record's colour quad rather than behind them. */ + uint8_t on_colour; + } set[XPSB_NSET_MAX]; + /* Temporaries the fragment program needs. Only the pixel data-master + * word reads this - a task holding more temporaries holds fewer + * pixels - and zero means the caller does not know, which leaves the + * task size bounded by the attribute space alone. */ + unsigned ntemps; + /* The program can discard. The vendor sets DOUTU1's punch-through + * field to phase 1 for exactly this - validate.c, beside the ISP pass + * type that already knows about it - and a punch-through object is + * shaded in two passes, so the task has to be told which it is. */ + unsigned uses_kill; + /* Which triangle corner a flat-shaded colour comes from, as 1 + the + * corner, or 0 to interpolate. It reaches the DOUTI FLATSHADE field of + * whichever issue carries the colour - the packed one, or the + * coordinate set it rides in under FIX_HW_BRN_25211 - because the + * MTE's own shade model only selects a base colour, and the routed + * form has none. */ + unsigned flatshade; + /* 1 + the set carrying the vertex colour under the FIX_HW_BRN_25211 + * routing, or 0 when the colour is on the base slot. SGX535 rev 1.2.1 + * carries that erratum (hwdefs/sgxerrata.h:785) and the vendor's + * workaround takes the colour off EURASIA_MTE_BASE and onto a texture + * coordinate set (opengles1/fftnlgles.c:900-975). */ + unsigned colour_set; +}; +/* Which base this program's frame uses: the captured slot while the issue + * list fits in it, the free block above the video constants when it does not. + * A pure function of the attribs, so the driver writing the unit state and + * the builder laying the segment out agree without passing it around. */ +unsigned xpsb_pri_pds_off(const struct xpsb_attribs *a); + + +/* Install an attribute configuration: the primary PDS program's issue list, + * the primary-attribute register count, the vertex record stride and its DMA + * descriptor, and the MTE words that tell the tiler the record's shape. + * Returns 0, or -1 when the configuration does not fit. */ +int xpsb_heap_set_attribs(uint32_t *h, const struct xpsb_attribs *a, + unsigned *pri_dwords); + + +/* Where each quantity lands in the primary attribute bank, in issue order: + * the packed colour first, then each set's iterated floats and each sampled + * unit's texel. Any of the out pointers may be NULL. set_base is XPSB_NSET_MAX + * entries and tex_base is XPSB_MAX_TEX - one per texture issue, since several + * units can sample one coordinate set. Bases past those are not reported. */ +void xpsb_attrib_bases(const struct xpsb_attribs *a, unsigned *colour, + unsigned *set_base, unsigned *tex_base); + +/* Floats in the record an attribute configuration describes. */ +unsigned xpsb_attrib_stride(const struct xpsb_attribs *a); +unsigned xpsb_attrib_set_base(const struct xpsb_attribs *a); +unsigned xpsb_attrib_tc_index(const struct xpsb_attribs *a, + unsigned set); + +/* The DOUTI TEXPROJ code (EURASIA_PDS_DOUTI_TEXPROJ_*: 0 none, 1 RHW, 2 T, + * 3 S) the given sampled unit's issue carries. */ +unsigned xpsb_attrib_texproj(const struct xpsb_attribs *a, unsigned unit); + +/* Which float of a set the TAG's projection divides the coordinate by, or -1 + * for a set it does not project. TEXPROJ_T divides by T, and the declared + * dimension says which float that is - the third of a UVT set, the fourth of + * a UVST one - so whoever fills the record must put the divisor there. */ +int xpsb_attrib_proj_float(const struct xpsb_attribs *a, unsigned set); + +/* Issues the primary program carries, and where each sampled unit's issue is. + * Returns the number of sampled units; issue_of may be NULL. */ +unsigned xpsb_attrib_issues(const struct xpsb_attribs *a, unsigned *nissue, + unsigned char *issue_of); + +/* The vertex USE program's emit count, which follows the same record. */ +void xpsb_usse_set_vtx_dwords(uint32_t *u, unsigned dwords); + +/* first_draw > 0 drops that many leading draw blocks and, with it, the same + * number of clearing rastgeom records. Returns the record count, terminator + * included, or 0 if first_draw leaves no user draw. The _n form also sizes the + * user draw's vertex record; the plain one is its ntex == 1 case. */ +unsigned xpsb_gen_draw_records_from(uint32_t *b, unsigned first_draw); +unsigned xpsb_gen_draw_records_n(uint32_t *b, unsigned first_draw, + unsigned ntex); +void xpsb_gen_rastgeom_from(uint32_t *b, int w, int h, unsigned first_draw); +unsigned xpsb_draw_cmd_off(unsigned first_draw); + +/* Four vertices of XPSB_VTX_STRIDE floats and six indices. Returns the index + * count; *nvtx, if given, receives the vertex count. */ +unsigned xpsb_gen_quad(float *vtx, uint16_t *idx, const struct xpsb_quad *q, + unsigned *nvtx); +unsigned xpsb_gen_quad_n(float *vtx, uint16_t *idx, const struct xpsb_quad *q, + unsigned ntex, unsigned *nvtx); + +/* Relocation emitters. Return the record count, or 0 for a configuration that + * cannot be built. out must have room for XPSB_MAX_TA_RELOCS records. */ +unsigned xpsb_gen_ta_relocs(struct xpsb_reloc *out, + const struct xpsb_reloc_cfg *c); +unsigned xpsb_gen_raster_relocs(struct xpsb_reloc *out, + const struct xpsb_reloc_cfg *c); + +/* Surface helpers. bpp returns 0 for a format the hardware has no code for; + * check returns -1 unless stride == bpp * ALIGN(w,32). */ +unsigned xpsb_format_bpp(uint32_t format); +/* The row pitch a linear surface is read at, in bytes; see the definition. */ +unsigned xpsb_linear_stride(uint32_t format, uint32_t width); +int xpsb_surface_check(const struct xpsb_surface_desc *s); +/* Twiddled-surface address arithmetic. The hardware reads a twiddled level as + * Morton order over the low min(log2w,log2h) bits - Y in the even bit of each + * pair, X in the odd - with the longer axis' remaining bits above. Levels are + * packed contiguously, offset(n) = cpp * sum of w*h below n, with no alignment + * between them. Both are gl-re/textures.md section 3.2 and 5.1. */ +uint32_t xpsb_twiddle_index(uint32_t x, uint32_t y, + uint32_t log2w, uint32_t log2h); +uint32_t xpsb_twiddle_level_offset(uint32_t w, uint32_t h, uint32_t level); +/* Twiddle one level from a linear source. Returns -1 on a bad size. */ +int xpsb_twiddle_level(void *dst, const void *src, uint32_t w, uint32_t h, + uint32_t src_stride, unsigned bpp); +int xpsb_untwiddle_level(void *dst, const void *src, uint32_t w, uint32_t h, + uint32_t dst_stride, unsigned bpp); + +int xpsb_tex_state(uint32_t *w0, uint32_t *w1, + const struct xpsb_surface_desc *s); +int xpsb_tex_state_mode(uint32_t *w0, uint32_t *w1, + const struct xpsb_surface_desc *s, int mode); +int xpsb_heap_set_dest(uint32_t *h, const struct xpsb_surface_desc *d); + +extern const struct xpsb_draw_desc xpsb_draw_descs[4]; + +extern const struct xpsb_fb_desc xpsb_frame_bufs[XPSB_FRAME_NBUF]; +extern const int xpsb_ta_list[XPSB_FRAME_NBUF]; +extern const int xpsb_raster_list[XPSB_RAS_LIST_LEN]; +extern const uint64_t xpsb_raster_vflags[XPSB_RAS_LIST_LEN]; +extern const uint64_t xpsb_raster_vmask[XPSB_RAS_LIST_LEN]; +extern const struct xpsb_reloc xpsb_ta_relocs[XPSB_NUM_TA_RELOCS]; +extern const struct xpsb_reloc xpsb_raster_relocs[XPSB_NUM_RAS_RELOCS]; + +#endif /* _XPSB_FRAME_H_ */ diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/xpsb_heap.c mesa-26.2.2/src/gallium/drivers/sgx/xpsb_heap.c --- mesa-26.2.2.orig/src/gallium/drivers/sgx/xpsb_heap.c 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/xpsb_heap.c 2026-09-08 10:57:36.684728009 +0200 @@ -0,0 +1,514 @@ +/* Synthesising the parameter heap - see xpsb_heap.h. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: GPL-2.0-only + */ +#include "xpsb_heap.h" +#include "xpsb_pds.h" +#include "xpsb_frame.h" + + +#include + +#define S1L XPSB_PDS_S1L +#define S1H XPSB_PDS_S1H +#define S2L XPSB_PDS_S2L +#define S2H XPSB_PDS_S2H +#define MOVS XPSB_MOVS +#define HALT XPSB_PDS_HALTW + +/* MOVSA and the conditional MOVS the vertex program issues. */ +#define MOVSA(src1, src2, s0, s1, s2, s3, dest) \ + XPSB_PDS_ENC(0, XPSB_PDS_TYPE_MOVSA, XPSB_PDS_CC_ALWAYS, \ + src1, src2, s0, s1, s2, s3, dest) +#define MOVS_IF0(src1, src2, s0, s1, s2, s3, dest) \ + XPSB_PDS_ENC(0, XPSB_PDS_TYPE_MOVS, XPSB_PDS_CC_IF0, \ + src1, src2, s0, s1, s2, s3, dest) + +/* Instructions outside the MOVS family, still carried as encoded words: their + * operand layouts are decoded by tools/isa-pds but have no encoder here yet. + * The disassembly is the comment, so these are instructions rather than + * anonymous heap dwords. */ +#define PDS_MUL_IR0_DS1 0x67800070u /* mul ds1[48], ir0.l, ds1[0].l */ +#define PDS_AND_IR1_DS1_0 0xcf820030u /* and ds0[48], ir1, ds1[0] */ +#define PDS_AND_IR1_DS1_2 0xcf820830u /* and ds0[48], ir1, ds1[2] */ +#define PDS_TSTZ_P0 0x87600000u /* tstz p0, ds0[48] */ +#define PDS_BRA_5 0x90000005u /* (p0) bra 0x5 */ +#define PDS_BRA_14 0x90000014u /* (p0) bra 0x14 */ + +static const uint32_t pds_halt[] = { HALT }; +static const uint32_t pds_use[] = { + MOVS(0, 0, S1L, S1H, S2L, S2L, XPSB_PDS_PORT_DOUTU), + HALT +}; +static const uint32_t pds_tex[] = { + MOVS(0, 0, S1L, S1H, S2L, S2L, XPSB_PDS_PORT_DOUTU), + MOVS(1, 0, S2H, S1L, S1H, S1H, XPSB_PDS_PORT_DOUTI), + MOVS(1, 1, S1L, S1H, S2L, S2H, XPSB_PDS_PORT_DOUTT), + HALT +}; +static const uint32_t pds_const[] = { + MOVS(0, 24, S1L, S1H, S1L, S1H, XPSB_PDS_PORT_DOUTD), + MOVS(1, 1, S1L, S1H, S2L, S2L, XPSB_PDS_PORT_DOUTU), + HALT +}; +static const uint32_t pds_vtx[] = { + PDS_MUL_IR0_DS1, + MOVSA(0, 24, S1L, S1H, S2L, S2L, XPSB_PDS_PORT_DOUTD), + MOVS_IF0(2, 0, S1L, S1H, S2H, S2H, XPSB_PDS_PORT_DOUTU), + HALT +}; +/* The two programs that test a state word before issuing: the clear path picks + * between them, which is why they branch rather than run straight through. */ +static const uint32_t pds_cleara[] = { + PDS_AND_IR1_DS1_0, PDS_TSTZ_P0, PDS_BRA_5, + MOVS(0, 0, S1L, S1H, S2H, S2H, XPSB_PDS_PORT_DOUTU), + HALT +}; +static const uint32_t pds_clearb[] = { + PDS_AND_IR1_DS1_2, PDS_TSTZ_P0, PDS_BRA_14, + MOVS(1, 1, S1L, S1H, S2L, S2H, XPSB_PDS_PORT_DOUTD), + 0x170086e0u, /* mov32 ds1[48], ds1[3] */ + 0xf7800b31u, /* shl ds0[49], ir0, 11 */ + 0xcf621031u, /* and ds0[49], ds0[49], ds1[4] */ + 0xc762c070u, /* or ds1[48], ds0[49], ds1[48] */ + 0xf7800b32u, /* shl ds0[50], ir0, 11 */ + 0xcf641432u, /* and ds0[50], ds0[50], ds1[5] */ + 0xc764c070u, /* or ds1[48], ds0[50], ds1[48] */ + 0xf7811933u, /* shl ds0[51], ir1, 25 */ + 0xcf661833u, /* and ds0[51], ds0[51], ds1[6] */ + 0xc766c070u, /* or ds1[48], ds0[51], ds1[48] */ + MOVS(2, 24, S1L, S1H, S2L, S2L, XPSB_PDS_PORT_DOUTU), + HALT +}; + +/* Still transcribed: everything synthesis has not reached yet. Categories + * leave this table as they are understood - render-target and texture surface + * descriptors, the PDS binding triplets, the state blocks, the viewport and + * size constants, the scene header. */ +const struct xpsb_heap_dw xpsb_heap_rest[] = { + { 0x00001, 0x804ca000 }, { 0x00010, 0x00000008 }, { 0x00012, 0x00000002 }, + { 0x00013, 0x10000000 }, { 0x00016, 0x08000000 }, + { 0x00038, 0x00000020 }, { 0x00050, 0x00000020 }, + { 0x00051, 0x0000f800 }, { 0x00052, 0x804ca000 }, { 0x00059, 0x00002000 }, + { 0x00090, 0x00000010 }, { 0x000a8, 0x00000020 }, { 0x000b1, 0x000054c5 }, + /* Measured inert: no single bit of either changes a pixel, nor does + * zeroing or filling both, on two different scenes. Kept because the + * synthesis reproduces the capture bit for bit, not because they are + * required - see work/heap-probe/README.md. */ + { 0x000b2, 0x01e00100 }, { 0x000b3, 0x0e000000 }, { 0x000b9, 0x04001000 }, + { 0x000ba, 0x00010000 }, { 0x000d8, 0x00000020 }, { 0x000d9, 0x0fc0aa00 }, + { 0x000da, 0x804ea000 }, { 0x01128, 0x00000010 }, { 0x01131, 0x00000045 }, + { 0x01132, 0x07f00100 }, { 0x01133, 0x02000000 }, { 0x01147, 0x00004f41 }, + { 0x01153, 0x0b001800 }, { 0x01154, 0x358637bd }, { 0x01155, 0x00000005 }, + { 0x10001, 0x80000000 }, +}; +const unsigned xpsb_heap_rest_n = + sizeof xpsb_heap_rest / sizeof xpsb_heap_rest[0]; + +const struct xpsb_heap_dw xpsb_heap_captured[] = { + { 0x00000, 0x003f8000 }, { 0x00001, 0x804ca000 }, { 0x00003, 0x0007f07f }, + { 0x00008, 0x00000225 }, { 0x0000a, 0x20010000 }, { 0x0000b, 0x80000003 }, + { 0x0000c, 0x00200025 }, { 0x00010, 0x00000008 }, { 0x00012, 0x00000002 }, + { 0x00013, 0x10000000 }, { 0x00014, 0x0007f800 }, + { 0x00015, 0x07f00000 }, { 0x00016, 0x08000000 }, + { 0x00018, 0xcf820030 }, { 0x00019, 0x87600000 }, + { 0x0001a, 0x90000005 }, { 0x0001b, 0x070003e5 }, { 0x0001c, 0xaf000000 }, + { 0x0001d, 0xcf820830 }, { 0x0001e, 0x87600000 }, { 0x0001f, 0x90000014 }, + { 0x00020, 0x07042363 }, { 0x00021, 0x170086e0 }, { 0x00022, 0xf7800b31 }, + { 0x00023, 0xcf621031 }, { 0x00024, 0xc762c070 }, { 0x00025, 0xf7800b32 }, + { 0x00026, 0xcf641432 }, { 0x00027, 0xc764c070 }, { 0x00028, 0xf7811933 }, + { 0x00029, 0xcf661833 }, { 0x0002a, 0xc766c070 }, { 0x0002b, 0x070b0345 }, + { 0x0002c, 0xaf000000 }, { 0x00030, 0x00000425 }, { 0x00038, 0x00000020 }, + { 0x0003c, 0x07000345 }, { 0x0003d, 0xaf000000 }, { 0x00040, 0xaf000000 }, + { 0x00048, 0x00100625 }, { 0x0004a, 0x001e0092 }, { 0x0004b, 0x6c07f07f }, + { 0x00050, 0x00000020 }, { 0x00051, 0x0000f800 }, { 0x00052, 0x804ca000 }, + { 0x00054, 0x07000345 }, { 0x00055, 0x070418a2 }, { 0x00056, 0x07042364 }, + { 0x00057, 0xaf000000 }, { 0x00058, 0xaf000000 }, { 0x00059, 0x00002000 }, + { 0x0005a, 0x00080008 }, { 0x00060, 0x20010164 }, { 0x00061, 0x80200001 }, + { 0x00062, 0x00200824 }, { 0x0006c, 0x07030223 }, { 0x0006d, 0x07042345 }, + { 0x0006e, 0xaf000000 }, { 0x00071, 0x3f800000 }, { 0x00072, 0x3f800000 }, + { 0x00073, 0x43000000 }, { 0x00075, 0x3f800000 }, { 0x00076, 0x3f800000 }, + { 0x00078, 0x43000000 }, { 0x00079, 0x3f800000 }, { 0x0007a, 0x3f800000 }, + { 0x0007b, 0x43000000 }, { 0x0007c, 0x43000000 }, { 0x0007d, 0x3f800000 }, + { 0x0007e, 0x3f800000 }, { 0x0007f, 0x00010000 }, { 0x00080, 0x00030002 }, + { 0x00081, 0x00010002 }, { 0x00088, 0x200101bc }, { 0x00089, 0x80600003 }, + { 0x0008c, 0x00200a24 }, { 0x00090, 0x00000010 }, { 0x00094, 0x67800070 }, + { 0x00095, 0x2f030343 }, { 0x00096, 0x030803e5 }, { 0x00097, 0xaf000000 }, + { 0x00098, 0xaf000000 }, { 0x000a0, 0x00000c25 }, { 0x000a8, 0x00000020 }, + { 0x000ac, 0x07000345 }, { 0x000ad, 0xaf000000 }, { 0x000b0, 0xaf000000 }, + { 0x000b1, 0x000054c5 }, { 0x000b2, 0x01e00100 }, { 0x000b3, 0x0e000000 }, + { 0x000b4, 0x0000102c }, { 0x000b5, 0x00030000 }, { 0x000b6, 0x0c001028 }, + { 0x000b9, 0x04001000 }, { 0x000ba, 0x00010000 }, { 0x000c0, 0x200102c4 }, + { 0x000c1, 0x8140000a }, { 0x000c2, 0x00200e24 }, { 0x000cc, 0x07030223 }, + { 0x000cd, 0x07042345 }, { 0x000ce, 0xaf000000 }, { 0x000d0, 0x00181025 }, + { 0x000d2, 0x03fe0090 }, { 0x000d3, 0x6c03f03f }, { 0x000d8, 0x00000020 }, + { 0x000d9, 0x0fc0aa00 }, { 0x000da, 0x804ea000 }, { 0x000dc, 0x07000345 }, + { 0x000dd, 0x070418a2 }, { 0x000de, 0x07042364 }, { 0x000df, 0xaf000000 }, + { 0x000e0, 0xaf000000 }, { 0x000e8, 0x40047800 }, { 0x000e9, 0x8140000a }, + { 0x000ec, 0x00201424 }, { 0x000f0, 0x0000002c }, { 0x000f4, 0x67800070 }, + { 0x000f5, 0x2f030343 }, { 0x000f6, 0x030803e5 }, { 0x000f7, 0xaf000000 }, + { 0x000f8, 0xaf000000 }, { 0x000f9, 0x00010000 }, { 0x000fa, 0x00010003 }, + { 0x000fb, 0x00030002 }, { 0x010fb, 0x3f800000 }, { 0x010fc, 0x3f800000 }, + { 0x010fd, 0x43000000 }, { 0x010ff, 0x3f800000 }, { 0x01100, 0x3f800000 }, + { 0x01102, 0x43000000 }, { 0x01103, 0x3f800000 }, { 0x01104, 0x3f800000 }, + { 0x01105, 0x43000000 }, { 0x01106, 0x43000000 }, { 0x01107, 0x3f800000 }, + { 0x01108, 0x3f800000 }, { 0x0110b, 0x3f800000 }, { 0x0110c, 0x3f800000 }, + { 0x0110d, 0x43000000 }, { 0x0110f, 0x3f800000 }, { 0x01110, 0x3f800000 }, + { 0x01112, 0x43000000 }, { 0x01113, 0x3f800000 }, { 0x01114, 0x3f800000 }, + { 0x01115, 0x43000000 }, { 0x01116, 0x43000000 }, { 0x01117, 0x3f800000 }, + { 0x01118, 0x3f800000 }, { 0x01119, 0x00010000 }, { 0x0111a, 0x00030002 }, + { 0x0111b, 0x00010002 }, { 0x01120, 0x200143e4 }, { 0x01121, 0x80600003 }, + { 0x01124, 0x00201624 }, { 0x01128, 0x00000010 }, { 0x0112c, 0x67800070 }, + { 0x0112d, 0x2f030343 }, { 0x0112e, 0x030803e5 }, { 0x0112f, 0xaf000000 }, + { 0x01130, 0xaf000000 }, { 0x01131, 0x00000045 }, { 0x01132, 0x07f00100 }, + { 0x01133, 0x02000000 }, { 0x01138, 0x200144c4 }, { 0x01139, 0x80a00005 }, + { 0x0113a, 0x00201824 }, { 0x01144, 0x07030223 }, { 0x01145, 0x07042345 }, + { 0x01146, 0xaf000000 }, { 0x01147, 0x00004f41 }, { 0x01148, 0x01d00000 }, + { 0x01149, 0x00001038 }, { 0x0114a, 0x30030002 }, { 0x0114b, 0x0c001034 }, + { 0x0114c, 0x42800000 }, { 0x0114d, 0x42800000 }, { 0x0114e, 0x42800000 }, + { 0x0114f, 0xc2800000 }, { 0x01150, 0x3f000000 }, { 0x01151, 0x3f000000 }, + { 0x01153, 0x0b001800 }, { 0x01154, 0x358637bd }, { 0x01155, 0x00000005 }, + { 0x01158, 0x2001451c }, { 0x01159, 0x81c0000e }, { 0x0115a, 0x00201a24 }, + { 0x01164, 0x07030223 }, { 0x01165, 0x07042345 }, { 0x01166, 0xaf000000 }, + { 0x10000, 0x031f8000 }, { 0x10001, 0x80000000 }, { 0x10002, 0x0005003c }, + { 0x10003, 0x000cf0bb }, { 0x10008, 0x67676767 }, { 0x1000a, 0x20050000 }, + { 0x1000b, 0x80000003 }, { 0x1000c, 0x67676767 }, { 0x10010, 0x00000008 }, + { 0x10012, 0x00000002 }, { 0x10013, 0x10000000 }, { 0x10014, 0x0007f800 }, + { 0x10015, 0x07f00000 }, { 0x10016, 0x08000000 }, { 0x10018, 0xcf820030 }, + { 0x10019, 0x87600000 }, { 0x1001a, 0x90000005 }, { 0x1001b, 0x070003e5 }, + { 0x1001c, 0xaf000000 }, { 0x1001d, 0xcf820830 }, { 0x1001e, 0x87600000 }, + { 0x1001f, 0x90000014 }, { 0x10020, 0x07042363 }, { 0x10021, 0x170086e0 }, + { 0x10022, 0xf7800b31 }, { 0x10023, 0xcf621031 }, { 0x10024, 0xc762c070 }, + { 0x10025, 0xf7800b32 }, { 0x10026, 0xcf641432 }, { 0x10027, 0xc764c070 }, + { 0x10028, 0xf7811933 }, { 0x10029, 0xcf661833 }, { 0x1002a, 0xc766c070 }, + { 0x1002b, 0x070b0345 }, { 0x1002c, 0xaf000000 }, { 0x10030, 0x67676767 }, + { 0x10032, 0x001e0092 }, { 0x10033, 0x6c07f07f }, { 0x10038, 0x00000020 }, + { 0x10039, 0x0000f800 }, { 0x1003a, 0x804ca000 }, { 0x1003c, 0x07000345 }, + { 0x1003d, 0x070418a2 }, { 0x1003e, 0x07042364 }, { 0x1003f, 0xaf000000 }, + { 0x10040, 0xaf000000 }, +}; +const unsigned xpsb_heap_captured_n = + sizeof xpsb_heap_captured / sizeof xpsb_heap_captured[0]; + +static const struct { const uint32_t *w; unsigned n; } pds_prog[] = { + { pds_halt, sizeof pds_halt / 4 }, + { pds_use, sizeof pds_use / 4 }, + { pds_tex, sizeof pds_tex / 4 }, + { pds_const, sizeof pds_const / 4 }, + { pds_vtx, sizeof pds_vtx / 4 }, + { pds_cleara, sizeof pds_cleara / 4 }, + { pds_clearb, sizeof pds_clearb / 4 }, +}; + +unsigned xpsb_heap_pds_emit(uint32_t *p, enum xpsb_heap_pds which) +{ + if ((unsigned)which >= XPSB_HEAP_PDS_COUNT) + return 0; + memcpy(p, pds_prog[which].w, pds_prog[which].n * 4); + return pds_prog[which].n; +} + +/* Each PDS program's data segment is loaded by a DOUTD: a source address and + * a control word. Seven of the nine are the ordinary contiguous form + * XPSB_DMA_CTL() builds, where the stride field comes out as bsize - 1; two + * ask for four dwords with a zero stride instead, so those carry their control + * word explicitly rather than being forced into the same helper. + * + * The vertex fetch's source is a placeholder: a relocation rewrites it to the + * vertex buffer, so what matters here is only that synthesis reproduces the + * word the relocation expects to find. */ +#define HEAPA(off) (XPSB_HEAP_ADDR + (off)) + +static const struct { uint32_t at, src; unsigned n; uint32_t ctl; } pds_desc[] = { + { 0x0000a, HEAPA(0x00000), 4, 0x80000003 }, + { 0x00060, HEAPA(0x00164), 2, 0 }, + { 0x00088, HEAPA(0x001bc), 4, 0 }, + { 0x000c0, HEAPA(0x002c4), 11, 0 }, + { 0x000e8, 0x40047800, 11, 0 }, /* relocated to the vertex buffer */ + { 0x01120, HEAPA(0x043e4), 4, 0 }, + { 0x01138, HEAPA(0x044c4), 6, 0 }, + { 0x01158, HEAPA(0x0451c), 15, 0 }, + { 0x1000a, HEAPA(0x40000), 4, 0x80000003 }, +}; + +/* The DOUTU payload word each state block carries: the USE program to run and + * how it is sequenced. Bits [18:8] are the program address >> 4, [7:4] a + * constant offset in 32 KiB units, and [21] the DMA dependency. Every one of + * the eight decodes with its address equal to the pre_add of the relocation + * that patches it, which is the cross-check that this reading is right. + * + * The low nibble is 4 everywhere except the first block, which uses 5; what it + * selects is NOT established, so it is carried per block rather than assumed. */ +#define DOUTU_W(exe, dep, low) \ + ((dep) | (2u << 4) | ((((uint32_t)(exe)) >> 4) << 8) | (low)) + +#define DOUTU_ITER 0x00080000u /* [19] wait on the iterators */ +#define DOUTU_TEX 0x00100000u /* [20] wait on the texture unit */ +#define DOUTU_DMA 0x00200000u /* [21] wait on the DMA */ + +static const struct { uint32_t at, exe, dep, low; } doutu_at[] = { + { 0x00008, 0x020, 0, 5 }, + { 0x0000c, 0x000, DOUTU_DMA, 5 }, + { 0x00030, 0x040, 0, 5 }, + { 0x00048, 0x060, DOUTU_TEX, 5 }, + { 0x00062, 0x080, DOUTU_DMA, 4 }, + { 0x0008c, 0x0a0, DOUTU_DMA, 4 }, + { 0x000a0, 0x0c0, 0, 5 }, + { 0x000c2, 0x0e0, DOUTU_DMA, 4 }, + { 0x000d0, 0x100, DOUTU_ITER | DOUTU_TEX, 5 }, + { 0x000ec, 0x140, DOUTU_DMA, 4 }, + { 0x01124, 0x160, DOUTU_DMA, 4 }, + { 0x0113a, 0x180, DOUTU_DMA, 4 }, + { 0x0115a, 0x1a0, DOUTU_DMA, 4 }, +}; + +/* Three DOUTU words the relocations overwrite outright. The capture holds a + * fill pattern there, which is the clearest possible evidence that nothing + * reads them before the relocation runs. */ +#define DOUTU_FILL 0x67676767u +static const uint32_t doutu_fill_at[] = { 0x10008, 0x1000c, 0x10030 }; + +/* A PDS binding triplet: the secondary program, a control word, the primary + * program. Each program is named by xpsb_pds_desc_word() - its address in bits + * 23:0 as a 16-byte unit, its data segment size in dwords in 31:24. + * + * The control word's secondary-attribute allocation (bits 24:18) is zero in + * both of these because this frame binds no secondary attributes, and bits + * 3:0 are the iterated varyings besides position. Bits 17:16 are 3 in both and + * bits 31:28 differ between them; neither field is decoded, so they are + * carried rather than computed. */ +static const struct { + uint32_t at, sec_off, sec_size, pri_off, pri_size, ctl; +} pds_bind[] = { + { 0x000b4, 0x2c0, 0, 0x280, 12, 0x00030000 }, + { 0x01149, 0x380, 0, 0x340, 12, 0x30030002 }, +}; + +/* The three full-screen quads and their index lists. + * + * Each is four vertices of (x, y, 1.0, 1.0) at the corners (0,0), (w,0), + * (0,h), (w,h) - sixteen floats - and the pair at heap+0x1bc and +0x4424 is + * followed by six 16-bit indices packed three to a dword. The quad at + * heap+0x43e4 has none of its own: heap+0x4424 begins immediately after its + * last vertex, so the two share the list that follows the second. + * + * Only the x and y of these were generated before; the z and w were 1.0 in the + * literal table and the index lists were literals too. */ +static void heap_quad(uint32_t *h, unsigned dw, float w, float hgt) +{ + static const float corner[4][2] = { {0,0}, {1,0}, {0,1}, {1,1} }; + const float one = 1.0f; + unsigned v; + + for (v = 0; v < 4; v++) { + float x = corner[v][0] * w, y = corner[v][1] * hgt; + + memcpy(&h[dw + v * 4 + 0], &x, 4); + memcpy(&h[dw + v * 4 + 1], &y, 4); + memcpy(&h[dw + v * 4 + 2], &one, 4); + memcpy(&h[dw + v * 4 + 3], &one, 4); + } +} + +static void heap_indices(uint32_t *h, unsigned dw, const uint16_t *idx) +{ + unsigned i; + + for (i = 0; i < 3; i++) + h[dw + i] = (uint32_t)idx[2 * i] | ((uint32_t)idx[2 * i + 1] << 16); +} + +static void heap_size_fields(uint32_t *h, int w, int hgt, unsigned sp, + uint32_t fmt) +{ + /* The two windings the heap carries: the full-screen quads use the + * first, the vertex-fetch descriptor at heap+0x3e4 the second. */ + static const uint16_t quad_idx[6] = { 0, 1, 2, 3, 2, 1 }; + static const uint16_t tri_idx[6] = { 0, 1, 3, 1, 2, 3 }; + float fw = (float)w, fh = (float)hgt; + float hw2 = fw * 0.5f, hh2 = fh * 0.5f, nh2 = -hh2; + uint32_t *vp; + + h[0x000] = (sp - 1u) << 15; + h[0x003] = ((unsigned)(hgt - 1) << 12) | (sp - 1u); + h[0x04b] = XPSB_SURF_FMT(fmt) | + ((unsigned)(w - 1) << 12) | (unsigned)(hgt - 1); + h[0x05a] = ((((unsigned)w + 15u) / 16u) << 16) | + (((unsigned)hgt + 15u) / 16u); + + heap_quad(h, 0x0006f, fw, fh); + heap_quad(h, 0x010f9, fw, fh); + heap_quad(h, 0x01109, fw, fh); + heap_indices(h, 0x0007f, quad_idx); + heap_indices(h, 0x01119, quad_idx); + heap_indices(h, 0x000f9, tri_idx); + + /* Present destination: origin, extent and surface size. The tile range + * that actually bounds the blit is in the raster stream. */ + h[0x10002] = ((unsigned)XPSB_PRESENT_Y << 12) | (unsigned)XPSB_PRESENT_X; + h[0x10003] = (((unsigned)(XPSB_PRESENT_Y + hgt) - 1u) << 12) | + ((unsigned)(XPSB_PRESENT_X + w) - 1u); + h[0x10033] = XPSB_SURF_FMT(fmt) | + ((unsigned)(w - 1) << 12) | (unsigned)(hgt - 1); + /* Group 8 of the state block: x centre and scale, y centre and scale. */ + vp = h + xpsb_heap_state_off(h, XPSB_STATE_VIEWPORT); + memcpy(&vp[0], &hw2, 4); + memcpy(&vp[1], &hw2, 4); + memcpy(&vp[2], &hh2, 4); + memcpy(&vp[3], &nh2, 4); +} + +/* The heap's default texture state: the sampler the texreplace object uses to + * read the render target back, and the sampler for the frame's own texture. + * Both come out of xpsb_tex_state_mode(), which is what binding a real texture + * calls too - so these are defaults rather than a separate encoding. */ +static void heap_tex_fields(uint32_t *h) +{ + struct xpsb_surface_desc s; + uint32_t w0, w1; + + memset(&s, 0, sizeof s); + s.format = XPSB_FMT_8888; + s.umode = 1; s.vmode = 1; + s.w = 64; s.h = 64; s.stride = 64 * 4; + + /* The load-back sampler, in the mode this driver emits. Its surface + * word is the render target's and heap_size_fields() writes it. */ + if (!xpsb_tex_state_mode(&w0, &w1, &s, XPSB_TEXCTL_XPSB)) { + h[0x0004a] = w0; + h[0x10032] = w0; + } + /* The frame's texture unit, in the captured mode. Binding a texture + * through xpsb_frame_set_texture() overwrites both words. */ + if (!xpsb_tex_state_mode(&w0, &w1, &s, XPSB_TEXCTL_CAPTURE)) { + h[XPSB_TEXCTL_OFF / 4] = w0; + h[XPSB_TEXSTATE_OFF / 4] = w1; + } +} + +/* The third scene block repeats three runs of the first verbatim. Emitting the + * copy from the original rather than listing the same words twice records that + * they are the same thing - whatever they turn out to mean. The blocks are not + * a uniform +0x10000 copy of each other, so only these three runs mirror. */ +static const struct { uint32_t dst, src, n; } mirror[] = { + { 0x10010, 0x00010, 1 }, /* the count word */ + { 0x10012, 0x00012, 5 }, /* the scene header run */ + { 0x10038, 0x00050, 3 }, /* the load-back state block */ +}; + +/* The remaining fields whose encoding is known. + * + * XPSB_PRESENT_STRIDE is the panel surface the present blit targets, which is + * why it is 1600 rather than the render size: (stride - 1) << 15 is the same + * surface-descriptor field heap_size_fields() writes for the render target. + * + * The per-draw ISP word is the captured default - compare function ALWAYS in + * bits 24:22 with depth write-disable in bit 20 - which xpsb_gen_heap() then + * rewrites for the depth and blend state the caller asked for. + * + * The vertex stride is the eleven-float record; xpsb_gen_vtxdma() overwrites it + * when a different number of sampled units is configured. */ +#define XPSB_PRESENT_STRIDE 1600u +#define ISP_FUNC_ALWAYS (7u << 22) +#define ISP_DEPTH_WRITE_OFF (1u << 20) + +static void heap_misc_fields(uint32_t *h) +{ + /* The two rasterised-extent controls: X at [17:12] of dword 0x14, Y at + * [24:20] of dword 0x15. Each set bit doubles the extent covered, from + * a base of 16 pixels - measured as cols = 16 * 2^popcount(field) and + * rows likewise, at four heights and two widths. Anything past the + * extent is left black rather than clipped or wrapped. + * + * All ones is what a driver wants and what the capture holds: it is the + * maximum, so one value serves every render size. The bits outside each + * field are set in the capture and inert, and are preserved. */ + h[0x014] = XPSB_EXT_X_REST | XPSB_EXT_X(0x3fu); + h[0x015] = XPSB_EXT_Y_REST | XPSB_EXT_Y(0x1fu); + + const float half = 0.5f; + uint32_t *vp = h + xpsb_heap_state_off(h, XPSB_STATE_VIEWPORT); + + h[0x10000] = (XPSB_PRESENT_STRIDE - 1u) << 15; + h[xpsb_heap_state_off(h, XPSB_STATE_ISP_A)] = + ISP_FUNC_ALWAYS | ISP_DEPTH_WRITE_OFF; + h[XPSB_VTXSTRIDE_DW] = 11u * 4u; + /* z centre and scale, the last pair of group 8 */ + memcpy(&vp[4], &half, 4); + memcpy(&vp[5], &half, 4); +} + +/* Where each program sits. The heap carries several copies of most of them, + * one per state block. */ +static const struct { uint32_t at; enum xpsb_heap_pds prog; } pds_at[] = { + { 0x00018, XPSB_HEAP_PDS_CLEARA }, { 0x0001d, XPSB_HEAP_PDS_CLEARB }, + { 0x0003c, XPSB_HEAP_PDS_USE }, { 0x00040, XPSB_HEAP_PDS_HALT }, + { 0x00054, XPSB_HEAP_PDS_TEX }, { 0x00058, XPSB_HEAP_PDS_HALT }, + { 0x0006c, XPSB_HEAP_PDS_CONST }, + { 0x00094, XPSB_HEAP_PDS_VTX }, { 0x00098, XPSB_HEAP_PDS_HALT }, + { 0x000ac, XPSB_HEAP_PDS_USE }, { 0x000b0, XPSB_HEAP_PDS_HALT }, + { 0x000cc, XPSB_HEAP_PDS_CONST }, + { 0x000dc, XPSB_HEAP_PDS_TEX }, { 0x000e0, XPSB_HEAP_PDS_HALT }, + { 0x000f4, XPSB_HEAP_PDS_VTX }, { 0x000f8, XPSB_HEAP_PDS_HALT }, + { 0x0112c, XPSB_HEAP_PDS_VTX }, { 0x01130, XPSB_HEAP_PDS_HALT }, + { 0x01144, XPSB_HEAP_PDS_CONST }, + { 0x01164, XPSB_HEAP_PDS_CONST }, + { 0x10018, XPSB_HEAP_PDS_CLEARA }, { 0x1001d, XPSB_HEAP_PDS_CLEARB }, + { 0x1003c, XPSB_HEAP_PDS_TEX }, { 0x10040, XPSB_HEAP_PDS_HALT }, +}; + +void xpsb_heap_synth(uint32_t *heap, const struct xpsb_heap_cfg *cfg) +{ + static const struct xpsb_heap_cfg captured = { 128, 128, 128, XPSB_FMT_8888 }; + unsigned i; + + if (!cfg) + cfg = &captured; + + for (i = 0; i < xpsb_heap_rest_n; i++) + heap[xpsb_heap_rest[i].at] = xpsb_heap_rest[i].v; + for (i = 0; i < sizeof pds_at / sizeof pds_at[0]; i++) + xpsb_heap_pds_emit(heap + pds_at[i].at, pds_at[i].prog); + for (i = 0; i < sizeof pds_desc / sizeof pds_desc[0]; i++) { + heap[pds_desc[i].at] = pds_desc[i].src; + heap[pds_desc[i].at + 1] = pds_desc[i].ctl ? pds_desc[i].ctl + : XPSB_DMA_CTL(pds_desc[i].n); + } + for (i = 0; i < sizeof doutu_at / sizeof doutu_at[0]; i++) + heap[doutu_at[i].at] = DOUTU_W(doutu_at[i].exe, doutu_at[i].dep, + doutu_at[i].low); + for (i = 0; i < sizeof doutu_fill_at / sizeof doutu_fill_at[0]; i++) + heap[doutu_fill_at[i]] = DOUTU_FILL; + for (i = 0; i < sizeof pds_bind / sizeof pds_bind[0]; i++) { + heap[pds_bind[i].at] = + xpsb_pds_desc_word(HEAPA(pds_bind[i].sec_off), + pds_bind[i].sec_size); + heap[pds_bind[i].at + 1] = pds_bind[i].ctl; + heap[pds_bind[i].at + 2] = + xpsb_pds_desc_word(HEAPA(pds_bind[i].pri_off), + pds_bind[i].pri_size); + } + heap_tex_fields(heap); + heap_misc_fields(heap); + for (i = 0; i < sizeof mirror / sizeof mirror[0]; i++) + memcpy(heap + mirror[i].dst, heap + mirror[i].src, + mirror[i].n * 4); + heap_size_fields(heap, cfg->w, cfg->h, cfg->stride, cfg->fmt); +} + +/* How many of the captured dwords synthesis now produces. Derived rather than + * counted by hand: the generators also write dwords the capture never listed, + * because the capture only recorded the non-zero ones, and counting those as + * progress would flatter the number. test_heap checks the whole heap either + * way, zeros included. */ +unsigned xpsb_heap_synthesised(void) +{ + return xpsb_heap_captured_n - xpsb_heap_rest_n; +} + +unsigned xpsb_heap_literals(void) +{ + return xpsb_heap_rest_n; +} diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/xpsb_heap.h mesa-26.2.2/src/gallium/drivers/sgx/xpsb_heap.h --- mesa-26.2.2.orig/src/gallium/drivers/sgx/xpsb_heap.h 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/xpsb_heap.h 2026-09-08 10:57:36.684742129 +0200 @@ -0,0 +1,88 @@ +/* Synthesising the parameter heap. + * + * The heap this driver submits started life as a transcribed capture: a table + * of literal dwords, patched afterwards by generators for the parts that were + * understood. That is enough to parameterise one known-good frame and not + * enough to build an arbitrary one, which is what a real driver needs. This + * file is where the literals are replaced by code, one category at a time. + * + * The rule is that synthesis must reproduce the capture exactly. Every + * category moved across is checked dword for dword against the transcription + * by test_heap, so a category is either fully understood or still a literal - + * never approximately right. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: GPL-2.0-only + */ +#ifndef _XPSB_HEAP_H_ +#define _XPSB_HEAP_H_ + +#include + +/* One heap dword and where it goes. */ +struct xpsb_heap_dw { uint32_t at; uint32_t v; }; + +/* The parameter heap exactly as captured, which synthesis has to reproduce. + * This is the reference test_heap diffs against; it shrinks to nothing only + * when every category has been moved into code. */ +extern const struct xpsb_heap_dw xpsb_heap_captured[]; +extern const unsigned xpsb_heap_captured_n; + +/* The part of it that is still a literal table rather than generated. */ +extern const struct xpsb_heap_dw xpsb_heap_rest[]; +extern const unsigned xpsb_heap_rest_n; + +/* The PDS programs the parameter heap carries. Six distinct ones, placed at + * fourteen offsets - the heap holds several copies of most of them. */ +enum xpsb_heap_pds { + XPSB_HEAP_PDS_HALT, /* halt on its own: the null secondary program */ + XPSB_HEAP_PDS_USE, /* movs doutu; halt */ + XPSB_HEAP_PDS_TEX, /* movs doutu; douti; doutt; halt - pixel, one unit */ + XPSB_HEAP_PDS_CONST, /* movs doutd; movs doutu; halt - the constant loader */ + XPSB_HEAP_PDS_VTX, /* mul; movsa doutd; movs doutu; halt - vertex fetch */ + XPSB_HEAP_PDS_CLEARA, /* the two conditional clear programs, which */ + XPSB_HEAP_PDS_CLEARB, /* branch on a state word before issuing */ + XPSB_HEAP_PDS_COUNT +}; + +/* What the heap is being built for. Passing NULL where a config is expected + * means the captured one - 128x128, 8888, stride 128 - which is what test_heap + * diffs against. */ +struct xpsb_heap_cfg { + int w, h; + unsigned stride; /* in pixels */ + uint32_t fmt; +}; + +/* Emit one program at *p; returns the dwords written. */ +unsigned xpsb_heap_pds_emit(uint32_t *p, enum xpsb_heap_pds which); + +/* Write the whole parameter heap. Reproduces the transcribed capture exactly; + * xpsb_heap_literals() reports how much of it is still a literal table, which + * is the number this file exists to drive to zero. */ +void xpsb_heap_synth(uint32_t *heap, const struct xpsb_heap_cfg *cfg); +unsigned xpsb_heap_literals(void); +unsigned xpsb_heap_synthesised(void); + + +/* The rasterised-extent controls, heap dwords 0x14 and 0x15. Each set bit in + * the field doubles the extent covered, from a base of 16 pixels: + * + * pixels = 16 * 2 ^ popcount(field) + * + * X is [17:12] of dword 0x14 and Y is [24:20] of dword 0x15, measured at four + * heights and two widths. Anything past the extent is left black rather than + * clipped or wrapped, so all ones - the maximum, and what the capture holds - + * is what a driver wants; one value then serves every render size. + * + * The remaining set bits in each dword are measured inert. They are kept so + * the synthesis still reproduces the capture bit for bit, not because they are + * known to do anything. + */ +#define XPSB_EXT_X(v) (((v) & 0x3fu) << 12) +#define XPSB_EXT_Y(v) (((v) & 0x1fu) << 20) +#define XPSB_EXT_X_REST 0x00040800u +#define XPSB_EXT_Y_REST 0x06000000u + +#endif diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/xpsb_hw.c mesa-26.2.2/src/gallium/drivers/sgx/xpsb_hw.c --- mesa-26.2.2.orig/src/gallium/drivers/sgx/xpsb_hw.c 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/xpsb_hw.c 2026-09-08 10:57:36.684758660 +0200 @@ -0,0 +1,759 @@ +/* Open replacement for Xpsb.so - SGX hardware layer. + * + * Transcribed from the disassembly of Xpsb_sgx_initialize, Xpsb_set_vopt, + * Xpsb_poll_reg_for_value, Xpsb_er_kick_wait_clear, Xpsb_closed_kick_ta, + * Xpsb_closed_kick_render, Xpsb_sgx_check_lockup, Xpsb_scene_info, + * Xpsb_ta_mem_info, Xpsb_ta_mem_load, Xpsb_scene_switch_fire and + * Xpsb_oom_abort. Every rule below is derived in README.md. + * + * The layer has no Xorg or DRM dependency so it can be exercised on its own. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#include +#include +#include +#include +#include + +#include "xpsb_hw.h" + +/* Kernel-side constants, from psb_drm.h / psb_reg.h. */ +#define PSB_SCENE_FLAG_DIRTY (1 << 0) +#define PSB_SCENE_FLAG_COMPLETE (1 << 1) +#define PSB_SCENE_FLAG_SETUP (1 << 2) +#define PSB_SCENE_FLAG_SETUP_ONLY (1 << 3) +#define PSB_TA_MEM_FLAG_TA (1 << 0) +#define PSB_TA_MEM_FLAG_RASTER (1 << 1) +#define PSB_TA_MEM_FLAG_HOSTA (1 << 2) +#define PSB_TA_MEM_FLAG_HOSTD (1 << 3) +#define PSB_TA_MEM_FLAG_INIT (1 << 4) +#define PSB_TA_MEM_FLAG_NEW_PT_OFFSET (1 << 5) +#define PSB_FIRE_FLAG_XHW_OOM (1 << 24) +#define PSB_SCENE_ENGINE_TA 0 +#define PSB_SCENE_ENGINE_RASTER 1 +#define PSB_RASTER_BLOCK 0 +#define PSB_RASTER 1 +#define PSB_RETURN 2 +#define PSB_TA 3 + +#define ALIGN4(v) (((v) + 3u) & ~3u) +#define KICK_TIMEOUT_US 100000u + +/* Round up to a power of two, matching the binary's loop exactly: the input is + * first masked to 28 bits, and zero maps to zero. */ +static uint32_t roundup_pow2(uint32_t v) +{ + uint32_t m, p; + + if (!v) + return 0; + m = v & 0x0fffffffu; + p = 1; + while (m > p) + p += p; + return p; +} + +/* Xpsb_set_vopt at 0x4af0. The core revision selects a set of errata + * workarounds; the key is 100 + minor*10 + maintenance and the enables + * cascade, so a lower revision turns on everything a higher one does. + * SGX535 rev 1.2.1 gives key 121, which enables nothing. */ +void xpsb_set_vopt(struct xpsb_hw *hw) +{ + uint32_t rev = xpsb_rd(hw, SGX_CORE_REVISION); + uint32_t key = 100u + ((rev >> 8) & 0xff) * 10u + (rev & 0xff); + int from, i; + + switch (key) { + case 107: from = 2; break; /* ctx+0x1c */ + case 108: from = 1; break; /* ctx+0x18 */ + case 109: from = 0; break; /* ctx+0x14 */ + case 111: + case 113: from = 13; break; /* ctx+0x48 */ + default: from = -1; break; + } + if (from >= 0) + for (i = from; i < (int)(sizeof hw->vopt / sizeof hw->vopt[0]); i++) + hw->vopt[i] = 1; + if (key == 111 || key == 113) + hw->vopt[31] = 1; /* ctx+0x90 */ +} + +/* Xpsb_sgx_initialize at 0x3820. */ +void xpsb_sgx_initialize(struct xpsb_hw *hw) +{ + hw->use_ctrl = 0xffff; + + xpsb_wrb(hw, SGX_BIF_CTRL2, 0xc07c); + xpsb_wr(hw, 0x13c, 0); + xpsb_wr(hw, SGX_DPM_LIMIT_C, 0); + xpsb_wrb(hw, SGX_DPM_LIMIT_D, 0); + xpsb_wr(hw, SGX_DPM_LIMIT_A, + hw->vopt[VOPT_DPM_LIMIT] ? 0x5021900u : 0x5188200u); + xpsb_wrb(hw, SGX_DPM_LIMIT_B, 0); + xpsb_wr(hw, SGX_DPM_THRESH_A, 0x44); + xpsb_wr(hw, SGX_DPM_THRESH_B, 0x0c); + if (hw->vopt[VOPT_CLKGATE]) + hw->use_ctrl = 0x100ffff; + xpsb_wrb(hw, SGX_USE_CTRL, hw->use_ctrl); + xpsb_wrb(hw, SGX_TSP_CTRL, 0x7c000); + xpsb_wr(hw, SGX_DPM_ALLOC_MODE, 0); + xpsb_wr(hw, SGX_MTE_CTRL, 0); +} + +/* Xpsb_sgx_uninit at 0x3810 is empty in the binary. */ +void xpsb_sgx_uninit(struct xpsb_hw *hw) +{ + (void)hw; +} + +/* Xpsb_poll_reg_for_value at 0x4ca0. The binary also accumulates a histogram + * of the two DPM state registers on every pass; that is pure instrumentation + * with no effect on the hardware and is not reproduced. */ +int xpsb_poll_reg(const struct xpsb_hw *hw, uint32_t off, uint32_t mask, + uint32_t val, uint32_t timeout_us) +{ + struct timeval start, now; + long long elapsed; + + gettimeofday(&start, NULL); + for (;;) { + if ((xpsb_rd(hw, off) & mask) == val) + return 0; + gettimeofday(&now, NULL); + elapsed = (long long)(now.tv_sec - start.tv_sec) * 1000000ll + + (now.tv_usec - start.tv_usec); + if (elapsed < 0) + elapsed = 1; + if ((unsigned long long)elapsed >= timeout_us) + return -EBUSY; + } +} + +/* Xpsb_er_kick_wait_clear at 0x4e10: optionally poke a register, wait for a + * status bit, acknowledge it, then wait for the acknowledge to take. A failed + * wait is reported but does not stop the acknowledge from being issued. */ +int xpsb_kick_wait_clear(const struct xpsb_hw *hw, int do_kick, + uint32_t kick_off, uint32_t kick_val, + uint32_t stat_off, uint32_t mask, uint32_t val, + uint32_t clear_off, uint32_t clear_val) +{ + int ret = 0; + + if (do_kick) + xpsb_wrb(hw, kick_off, kick_val); + + if (xpsb_poll_reg(hw, stat_off, mask, val, KICK_TIMEOUT_US)) + ret = -EBUSY; + + xpsb_wrb(hw, clear_off, clear_val); + (void)xpsb_poll_reg(hw, stat_off, mask, 0, KICK_TIMEOUT_US); + return ret; +} + +/* Xpsb_closed_kick_ta at 0x4f10. */ +int xpsb_kick_ta(const struct xpsb_hw *hw) +{ + xpsb_wr(hw, SGX_BIF_TA_FLUSH, 0x30000000); + xpsb_kick_wait_clear(hw, 1, SGX_TA_FLUSH, 1, + SGX_EVENT_STATUS3, 0x44, 0x44, + SGX_EVENT_HOST_CLEAR3, 0x44); + xpsb_kick_wait_clear(hw, 1, SGX_TA_FLUSH2, 1, + SGX_EVENT_STATUS3, 1, 1, + SGX_EVENT_HOST_CLEAR3, 1); + xpsb_kick_wait_clear(hw, 1, SGX_USE_CTRL, hw->use_ctrl | 0x10000000, + SGX_EVENT_STATUS, SGX_EV_KICK_DONE, SGX_EV_KICK_DONE, + SGX_EVENT_HOST_CLEAR, SGX_EV_KICK_DONE); + xpsb_wr(hw, SGX_PDS_CTRL, 1); + xpsb_wrb(hw, SGX_TA_KICK, 1); + return 0; +} + +/* Xpsb_closed_kick_render at 0x5040: the same preamble, a different kick. */ +int xpsb_kick_render(const struct xpsb_hw *hw) +{ + xpsb_kick_wait_clear(hw, 1, SGX_TA_FLUSH, 1, + SGX_EVENT_STATUS3, 0x44, 0x44, + SGX_EVENT_HOST_CLEAR3, 0x44); + xpsb_kick_wait_clear(hw, 1, SGX_TA_FLUSH2, 1, + SGX_EVENT_STATUS3, 1, 1, + SGX_EVENT_HOST_CLEAR3, 1); + xpsb_kick_wait_clear(hw, 1, SGX_USE_CTRL, hw->use_ctrl | 0x10000000, + SGX_EVENT_STATUS, SGX_EV_KICK_DONE, SGX_EV_KICK_DONE, + SGX_EVENT_HOST_CLEAR, SGX_EV_KICK_DONE); + xpsb_wr(hw, SGX_3D_CTRL, 1); + xpsb_wr(hw, SGX_PDS_CTRL, 1); + xpsb_wrb(hw, SGX_3D_KICK, 1); + return 0; +} + +/* Xpsb_sgx_check_lockup at 0x38f0: XOR eleven progress registers into a + * signature and sample it three times. Three identical samples, and a match + * against the previous call, means nothing advanced. */ +int xpsb_check_lockup(struct xpsb_hw *hw, uint32_t *value) +{ + uint32_t prev = hw->last_sig; + uint32_t sig = 0; + int i; + + for (i = 0; i < 3; i++) { + sig = xpsb_rd(hw, SGX_ISP_SIG_BASE + 0x18) ^ + xpsb_rd(hw, SGX_ISP_SIG_BASE + 0x14) ^ + xpsb_rd(hw, SGX_ISP_SIG_BASE + 0x00) ^ + xpsb_rd(hw, SGX_ISP_SIG_BASE + 0x04) ^ + xpsb_rd(hw, SGX_ISP_SIG_BASE + 0x08) ^ + xpsb_rd(hw, SGX_ISP_SIG_BASE + 0x0c) ^ + xpsb_rd(hw, SGX_TA_SIG_BASE + 0x00) ^ + xpsb_rd(hw, SGX_TA_SIG_BASE + 0x04) ^ + xpsb_rd(hw, SGX_TA_SIG_BASE + 0x08) ^ + xpsb_rd(hw, SGX_TA_SIG_BASE + 0x0c) ^ + (xpsb_rd(hw, SGX_MISC_SIG) & 0xffff); + if (sig != prev) { + hw->last_sig = sig; + *value = 0; + return 0; + } + } + *value = 1; + return 1; +} + +/* Xpsb_scene_info at 0x3a40. + * + * Tiles are 16x16 pixels and are grouped into macro tiles. Normally a macro + * tile is half the tile grid rounded up to a multiple of four; past 2047 + * pixels the packed 10-bit lanes would overflow, so the grouping switches to + * quarters and a three-lane encoding. The whole reply is a cookie the kernel + * only stores and hands back, plus the buffer size and the page range it must + * zero before the first use. */ +int xpsb_scene_info(const struct xpsb_hw *hw, uint32_t w, uint32_t h, + uint32_t *cookie, uint32_t *size, + uint32_t *clear_p_start, uint32_t *clear_num_pages) +{ + uint32_t tx = (w + 15u) >> 4, ty = (h + 15u) >> 4; + uint32_t mx = ALIGN4((tx + 1u) >> 1), my = ALIGN4((ty + 1u) >> 1); + uint32_t ex, ey, upt, px, py, pow2, base; + + if ((w > 0x7ff || h > 0x7ff) && !hw->vopt[VOPT_LARGE_SCENE]) { + cookie[0] = 0x80000000; + upt = 4; + mx = ALIGN4((tx + 3u) >> 2); + my = ALIGN4((ty + 3u) >> 2); + ex = mx; + ey = my; + cookie[2] = 0x80000000u | (mx << 22) | (mx << 13) | (mx * 3u); + cookie[3] = (my << 22) | (my << 13) | (my * 3u); + } else { + cookie[0] = 0; + upt = 2; + if (hw->vopt[VOPT_WIDE_MTILE]) { + ex = ALIGN4(tx * 2u); + ey = ALIGN4(ty * 2u); + if (ex > 0xff) + ex = 0xff; + if (ey > 0xff) + ey = 0xff; + } else { + ex = mx; + ey = my; + } + cookie[2] = (ex << 22) | (ex << 12) | ex; + cookie[3] = (ey << 22) | (ey << 12) | ey; + } + + cookie[1] = mx * my; + cookie[4] = ex * ey; + cookie[5] = ((ty - 1u) << 12) | (tx - 1u); + cookie[6] = ((h - 1u) << 12) | (w - 1u); + cookie[8] = 0; + + px = upt * mx; + py = upt * my; + pow2 = roundup_pow2(px); + if (roundup_pow2(py) > pow2) + pow2 = roundup_pow2(py); + cookie[7] = ((pow2 * pow2 * 4u) + 0xfffu) & ~0xfffu; + + base = cookie[7] + px * py * 12u; + cookie[9] = base; + cookie[10] = base + 0x50; + cookie[11] = base + 0x50 + 0x90; + cookie[12] = 0; + cookie[13] = 0; + cookie[14] = 0; + + *size = base + 0x50 + 0x90 + 0x40; + *clear_p_start = cookie[8] >> 12; + *clear_num_pages = ((cookie[7] + 0xfffu) >> 12) - *clear_p_start; + return 0; +} + +/* Xpsb_ta_mem_info at 0x3d70. This only validates: the kernel discards the + * size we return (psb_scene.c:462) and allocates drm_psb_ta_mem_size instead, + * 32 MB by default. So the responder sets a floor, it does not choose the + * heap, and an undersized request is the one thing it can refuse. */ +int xpsb_ta_mem_info(uint32_t pages, uint32_t *cookie, uint32_t *size) +{ + uint32_t avail; + + *size = 0x620000; + if (pages <= 0x61f) + return -ENOMEM; + + avail = pages - 0x600; + cookie[2] = pages; + cookie[3] = avail; + cookie[4] = avail; + cookie[5] = avail; + cookie[6] = (avail >= 0x41) ? (pages - 0x640) : 0; + memset(&cookie[7], 0, 6 * sizeof(uint32_t)); + return 0; +} + +/* Xpsb_ta_mem_load at 0x3df0: bind the parameter heap to any subset of the + * four DPM clients, then wait for each load to be acknowledged. */ +int xpsb_ta_mem_load(struct xpsb_hw *hw, uint32_t pt_offset, + uint32_t param_offset, uint32_t flags, uint32_t *cookie) +{ + uint32_t kick = 0, span, tail; + int ret; + + if (flags & (PSB_TA_MEM_FLAG_INIT | PSB_TA_MEM_FLAG_NEW_PT_OFFSET)) + cookie[0] = pt_offset; + + if (flags & PSB_TA_MEM_FLAG_INIT) { + uint32_t a = param_offset & 0x0fffffffu; + + cookie[1] = a; + cookie[10] = a >> 12; + cookie[11] = cookie[2] + (a >> 12) - 1u; + cookie[12] = cookie[11] - 1u; + } + + span = (cookie[11] << 16) | cookie[10]; + tail = cookie[12] << 16; + + if (flags & PSB_TA_MEM_FLAG_TA) { + kick |= 1; + xpsb_wr(hw, SGX_DPM_TA_BASE, cookie[0]); + xpsb_wr(hw, SGX_DPM_TA_BASE + 4, span); + xpsb_wr(hw, 0x648, tail); + xpsb_wr(hw, SGX_DPM_TA_GLOBAL, (cookie[8] | cookie[7]) & 0xffff); + xpsb_wr(hw, SGX_DPM_TA_LOCAL_A, cookie[4] & 0xffff); + xpsb_wr(hw, SGX_DPM_TA_LOCAL_B, cookie[5] & 0xffff); + xpsb_wr(hw, SGX_DPM_TA_LOCAL_C, cookie[3] & 0xffff); + xpsb_wr(hw, SGX_DPM_TA_TAIL, cookie[6] & 0xffff); + xpsb_wr(hw, SGX_DPM_LOAD_TA, 1); + } + if (flags & PSB_TA_MEM_FLAG_RASTER) { + kick |= 2; + xpsb_wr(hw, SGX_DPM_RASTER_BASE, cookie[0]); + xpsb_wr(hw, SGX_DPM_RASTER_BASE + 4, span); + xpsb_wr(hw, 0x64c, tail); + xpsb_wr(hw, SGX_DPM_RASTER_GLOBAL, + (cookie[8] | cookie[7]) & 0xffff); + xpsb_wr(hw, SGX_DPM_LOAD_RASTER, 1); + } + if (flags & PSB_TA_MEM_FLAG_HOSTA) { + kick |= 4; + xpsb_wr(hw, SGX_DPM_HOSTA_BASE, cookie[0]); + xpsb_wr(hw, SGX_DPM_HOSTA_BASE + 4, span); + xpsb_wr(hw, 0x654, tail); + xpsb_wr(hw, SGX_DPM_LOAD_HOSTA, 1); + } + if (flags & PSB_TA_MEM_FLAG_HOSTD) { + kick |= 4; /* the binary reuses bit 2 here */ + xpsb_wr(hw, SGX_DPM_HOSTD_BASE, cookie[0]); + xpsb_wr(hw, SGX_DPM_HOSTD_BASE + 4, span); + xpsb_wr(hw, 0x650, tail); + xpsb_wr(hw, SGX_DPM_LOAD_HOSTD, 1); + } + + ret = xpsb_kick_wait_clear(hw, 0, 0, 0, SGX_EVENT_STATUS2, kick, kick, + SGX_EVENT_HOST_CLEAR2, kick); + if (ret || !(flags & PSB_TA_MEM_FLAG_INIT)) + return ret; + + return xpsb_kick_wait_clear(hw, 1, SGX_DPM_LOAD_KICK, 1, + SGX_EVENT_STATUS, SGX_EV_TA_LOAD, + SGX_EV_TA_LOAD, SGX_EVENT_HOST_CLEAR, + SGX_EV_TA_LOAD); +} + +/* Mirror the DPM allocation mode into the register only when it changes, + * which is what the binary does at every one of its five update sites. */ +static void set_alloc_mode(struct xpsb_hw *hw, uint32_t v) +{ + if (v == hw->alloc_mode) + return; + hw->alloc_mode = v; + xpsb_wr(hw, SGX_DPM_ALLOC_MODE, v); +} + +/* Program the macro-tile geometry from a scene cookie. Shared by the clean + * TA setup and by the fire helper. Returns the region base it consumed. */ +static uint32_t emit_scene_setup(const struct xpsb_hw *hw, + const uint32_t *cookie, uint32_t offset) +{ + xpsb_wr(hw, SGX_TA_MTILE_X, cookie[2]); + xpsb_wr(hw, SGX_TA_MTILE_Y, cookie[3]); + xpsb_wr(hw, SGX_TA_MTILE_STRIDE, cookie[4]); + xpsb_wr(hw, SGX_TA_MTILE_SIZE, cookie[5]); + xpsb_wr(hw, SGX_TA_PIXEL_EXTENT, cookie[6]); + offset += cookie[7]; + xpsb_wr(hw, SGX_TA_REGION_BASE, offset); + return offset; +} + +/* The unnamed helper at 0x4030, reached from both fire paths. It binds the + * scene, and on a raster fire decides whether this is a normal render or the + * second half of an out-of-memory partial render. */ +static int scene_fire_common(struct xpsb_hw *hw, uint32_t *cookie, + int is_raster, uint32_t offset, uint32_t flags, + const uint32_t *oom_cmds, uint32_t num_oom_cmds, + uint32_t *rca) +{ + uint32_t wait_ev = 0, dpm_state = 0, extra = 0, mask, ctx_val; + uint32_t i, n; + + xpsb_wr(hw, SGX_DPM_ZLS_ENABLE, 0); + + if ((flags & (PSB_SCENE_FLAG_DIRTY | PSB_SCENE_FLAG_SETUP_ONLY)) == + PSB_SCENE_FLAG_DIRTY) { + xpsb_wr(hw, SGX_DPM_PAGE_TABLE, offset + cookie[10]); + xpsb_wr(hw, SGX_DPM_STATE_TABLE, offset + cookie[11]); + xpsb_wr(hw, SGX_DPM_REQUEST, 4); + xpsb_wr(hw, SGX_DPM_REQUEST2, 4); + wait_ev = 0x120; + if (!(flags & PSB_SCENE_FLAG_COMPLETE)) { + xpsb_wr(hw, SGX_TA_PARAM_BASE, offset + cookie[8]); + xpsb_wr(hw, SGX_TA_ZLS_BASE, offset + cookie[9]); + xpsb_wr(hw, SGX_TA_CTRL2, 2); + xpsb_wrb(hw, SGX_TA_START, 0x80000000); + wait_ev = 0x200920; + } + } + + if (!(flags & PSB_SCENE_FLAG_SETUP)) + goto out; + + if (!is_raster) { + emit_scene_setup(hw, cookie, offset); + goto out; + } + if (is_raster != 1) + goto out; + + *rca = PSB_RETURN; + if (cookie[14] == 0) + goto fire; + + if (getenv("XPSB_OOM_DUMP")) + fprintf(stderr, " scene_fire raster: cookie[14]=%u " + "cookie[13]=%08x noom=%u\n", cookie[14], cookie[13], + num_oom_cmds); + if (!(cookie[13] & 0x80000000u) && num_oom_cmds != 0) { + /* First pass of an OOM recovery: render what the tile arrays + * already hold, then come back for the rest. */ + cookie[13] |= 0x80000000u; + xpsb_wr(hw, SGX_DPM_PARTIAL, 0); + xpsb_wr(hw, SGX_DPM_CONTEXT, 0); + *rca = PSB_RASTER; + xpsb_wr(hw, SGX_3D_STATUS, 3); + goto tail; + } + + cookie[13] |= 0x80000000u; + dpm_state = xpsb_rd(hw, SGX_DPM_STATE); + if ((dpm_state & 0x30000) == 0x30000) + xpsb_wr(hw, SGX_BIF_3D_FLUSH, 0x30000000); + xpsb_wr(hw, SGX_ISP_CTRL, xpsb_rd(hw, SGX_ISP_CTRL) & ~0x100u); + + if (cookie[14] == 2) { + if (getenv("XPSB_OOM_DUMP")) + fprintf(stderr, " final render: cookie[13]=%08x, " + "oom cmds %s\n", cookie[13], + (cookie[13] & 0xffff) ? "APPLIED" : "SKIPPED"); + if ((cookie[13] & 0xffff) != 0) { + dpm_state |= 6; + xpsb_wr(hw, SGX_DPM_STATE, dpm_state); + n = num_oom_cmds >> 1; + for (i = 0; i < n; i++) + xpsb_wr(hw, oom_cmds[2 * i], oom_cmds[2 * i + 1]); + } + cookie[14] = 0; + cookie[13] = 0; + *rca = PSB_RETURN; + goto fire; + } + + /* Drain the three DPM phases, swapping the allocation context in the + * middle, so the partially built scene can be rasterised on its own. */ + ctx_val = (hw->alloc_mode & ~3u) | ((hw->alloc_mode & 4) ? 1u : 2u); + set_alloc_mode(hw, ctx_val); + xpsb_wr(hw, SGX_DPM_ZLS_ENABLE, ((cookie[15] >> 4) ^ 1u) & 1u); + xpsb_wr(hw, SGX_DPM_PAGE_TABLE, offset + cookie[10]); + + xpsb_wr(hw, SGX_DPM_REQUEST, 1); + if (xpsb_kick_wait_clear(hw, 0, 0, 0, SGX_EVENT_STATUS, SGX_EV_DPM_A, + SGX_EV_DPM_A, SGX_EVENT_HOST_CLEAR, + SGX_EV_DPM_A)) + goto fail; + xpsb_wr(hw, SGX_DPM_REQUEST, 2); + if (xpsb_kick_wait_clear(hw, 0, 0, 0, SGX_EVENT_STATUS, SGX_EV_DPM_C, + SGX_EV_DPM_C, SGX_EVENT_HOST_CLEAR, + SGX_EV_DPM_C)) + goto fail; + + set_alloc_mode(hw, ctx_val ^ 1u); + xpsb_wr(hw, SGX_DPM_REQUEST, 4); + if (xpsb_kick_wait_clear(hw, 0, 0, 0, SGX_EVENT_STATUS, SGX_EV_DPM_B, + SGX_EV_DPM_B, SGX_EVENT_HOST_CLEAR, + SGX_EV_DPM_B)) { + xpsb_wr(hw, SGX_DPM_ZLS_ENABLE, 0); + return -EINVAL; + } + xpsb_wr(hw, SGX_DPM_ZLS_ENABLE, 0); + + *rca = PSB_TA; + cookie[14] = 2; + if (cookie[15] & 0x10) { + dpm_state |= 6; + mask = 0xffff; + extra = 0; + ctx_val = 3; + } else { + uint32_t slot = cookie[15] & 0xf; + + mask = 1u << slot; + extra = slot * cookie[1] * 12u; + ctx_val = 2; + } + if (getenv("XPSB_OOM_DUMP")) + fprintf(stderr, " partial render: cookie[15]=%08x mask=%04x " + "cookie[13]=%08x -> oom cmds %s, extra=%u\n", + cookie[15], mask, cookie[13], + (cookie[13] & mask) ? "APPLIED" : "SKIPPED (first " + "time for this macro tile)", extra); + if (cookie[13] & mask) { + n = num_oom_cmds >> 1; + for (i = 0; i < n; i++) + xpsb_wr(hw, oom_cmds[2 * i], oom_cmds[2 * i + 1]); + dpm_state |= 6; + } else { + cookie[13] |= mask; + } + dpm_state |= 0x90; + xpsb_wr(hw, SGX_DPM_STATE, dpm_state); + xpsb_wr(hw, SGX_DPM_CONTEXT, ctx_val); + xpsb_wr(hw, SGX_DPM_PARTIAL, 1); + goto tail; + +fire: + xpsb_wr(hw, SGX_DPM_CONTEXT, 3); + xpsb_wr(hw, SGX_DPM_PARTIAL, 0); + extra = 0; +tail: + offset += cookie[7]; + xpsb_wr(hw, SGX_3D_PARAM_BASE, offset + extra); +out: + if (!wait_ev) + return 0; + return xpsb_kick_wait_clear(hw, 0, 0, 0, SGX_EVENT_STATUS, wait_ev, + wait_ev, SGX_EVENT_HOST_CLEAR, wait_ev); +fail: + xpsb_wr(hw, SGX_DPM_ZLS_ENABLE, 0); + return -EINVAL; +} + +/* The out-of-memory branch of Xpsb_scene_switch_fire at 0x47b8. This is the + * tiler *resume*, phase five of the partial render, not the handler the + * kernel names - that is Xpsb_oom_abort. CR_DPM_PAGE_TABLE is pointed at the + * private table (hw->oom_page_table), whose CPU mapping is hw->regions; the + * DPM rewrites that table once per allocation context, so the high halves of + * the per-context tile counters are saved across the swap and put back. + * Whether any headroom came back is recorded in cookie[13] bit 30, which is + * what stops the next OOM from livelocking. */ +/* Nothing in Xpsb.so ever writes the region records - the DPM produces them - + * so whether the buffer hw->regions points at is the one the hardware is + * actually filling can only be answered by looking. XPSB_OOM_DUMP=1 prints the + * seventeen records at each step of the recovery. */ +static void oom_dump(const struct xpsb_hw *hw, const char *when) +{ + unsigned i, nz = 0; + + if (!hw->regions || !getenv("XPSB_OOM_DUMP")) + return; + fprintf(stderr, " regions %-18s", when); + for (i = 0; i < 17; i++) { + if (hw->regions[i][0] || hw->regions[i][1]) + nz++; + if (i < 6) + fprintf(stderr, " %08x:%08x", hw->regions[i][0], + hw->regions[i][1]); + } + fprintf(stderr, " (%u of 17 non-zero)\n", nz); +} + +static int xpsb_oom_partial(struct xpsb_hw *hw, uint32_t *cookie) +{ + uint32_t save[16] = { 0 }, n, i, slot; + int save_all = (cookie[15] & 0x10) != 0; + + xpsb_wr(hw, SGX_DPM_ZLS_ENABLE, ((cookie[15] >> 4) ^ 1u) & 1u); + set_alloc_mode(hw, (hw->alloc_mode & ~1u) | + ((hw->alloc_mode >> 1) & 1u)); + xpsb_wr(hw, SGX_DPM_PAGE_TABLE, hw->oom_page_table); + oom_dump(hw, "on entry"); + xpsb_wr(hw, SGX_DPM_REQUEST, 1); + if (xpsb_kick_wait_clear(hw, 0, 0, 0, SGX_EVENT_STATUS, SGX_EV_DPM_A, + SGX_EV_DPM_A, SGX_EVENT_HOST_CLEAR, + SGX_EV_DPM_A)) + goto fail; + + oom_dump(hw, "after request 1"); + slot = cookie[15] & 0xf; + n = save_all ? ((cookie[0] == 0) ? 4u : 16u) : 0u; + if (hw->regions) { + if (save_all) + for (i = 0; i < n; i++) + save[i] = hw->regions[i][1] & 0xffff0000u; + else + save[slot] = hw->regions[slot][1] & 0xffff0000u; + } + + if (getenv("XPSB_OOM_DUMP")) + fprintf(stderr, " save_all=%d slot=%u n=%u cookie[15]=%08x " + "saved=%08x\n", save_all, slot, n, cookie[15], + save[save_all ? 0 : slot]); + oom_dump(hw, "after save"); + set_alloc_mode(hw, (hw->alloc_mode & ~1u) | + ((hw->alloc_mode >> 2) & 1u)); + xpsb_wr(hw, SGX_DPM_REQUEST, 1); + if (xpsb_kick_wait_clear(hw, 0, 0, 0, SGX_EVENT_STATUS, SGX_EV_DPM_A, + SGX_EV_DPM_A, SGX_EVENT_HOST_CLEAR, + SGX_EV_DPM_A)) + goto fail; + + oom_dump(hw, "after request 2"); + if (hw->regions) { + if (save_all) + for (i = 0; i < n; i++) + hw->regions[i][1] = + (hw->regions[i][1] & 0xffff) | save[i]; + else + hw->regions[slot][1] = + (hw->regions[slot][1] & 0xffff) | save[slot]; + } + + oom_dump(hw, "after restore"); + xpsb_wr(hw, SGX_DPM_REQUEST, 4); + if (xpsb_kick_wait_clear(hw, 0, 0, 0, SGX_EVENT_STATUS, SGX_EV_DPM_B, + SGX_EV_DPM_B, SGX_EVENT_HOST_CLEAR, + SGX_EV_DPM_B)) + goto fail; + xpsb_wr(hw, SGX_DPM_REQUEST2, 2); + if (xpsb_kick_wait_clear(hw, 0, 0, 0, SGX_EVENT_STATUS, SGX_EV_DPM_D, + SGX_EV_DPM_D, SGX_EVENT_HOST_CLEAR, + SGX_EV_DPM_D)) + goto fail; + + xpsb_wr(hw, SGX_DPM_ZLS_ENABLE, 0); + if (((xpsb_rd(hw, SGX_DPM_TA_LOCAL_B) & 0xffff) - + (xpsb_rd(hw, SGX_DPM_FREE) >> 16)) <= 0x1f) + cookie[13] |= 0x40000000u; + else + cookie[13] &= ~0x40000000u; + xpsb_wrb(hw, SGX_DPM_MODE, 1); + return 0; + +fail: + xpsb_wr(hw, SGX_DPM_ZLS_ENABLE, 0); + return 0; +} + +/* Xpsb_scene_switch_fire at 0x4550. */ +int xpsb_scene_switch_fire(struct xpsb_hw *hw, uint32_t fire_flags, + uint32_t hw_context, uint32_t *cookie, + uint32_t offset, uint32_t engine, uint32_t flags, + const uint32_t *oom_cmds, uint32_t num_oom_cmds, + uint32_t *rca) +{ + uint32_t ctx_bit = (hw_context == 1) ? 1u : 0u; + uint32_t base = (hw->alloc_mode & ~1u) | ctx_bit; + + if (engine == PSB_SCENE_ENGINE_RASTER) { + set_alloc_mode(hw, (base & ~2u) | (ctx_bit << 1)); + if ((flags & (PSB_SCENE_FLAG_DIRTY | PSB_SCENE_FLAG_COMPLETE)) != + (PSB_SCENE_FLAG_DIRTY | PSB_SCENE_FLAG_COMPLETE)) + return 0; + if (scene_fire_common(hw, cookie, 1, offset, flags, + oom_cmds, num_oom_cmds, rca) == -EINVAL) + return 0; + xpsb_kick_render(hw); + return 0; + } + if (engine != PSB_SCENE_ENGINE_TA) + return 0; + + set_alloc_mode(hw, (base & ~4u) | (ctx_bit << 2)); + + if (fire_flags & PSB_FIRE_FLAG_XHW_OOM) + return xpsb_oom_partial(hw, cookie); + + if (flags & PSB_SCENE_FLAG_DIRTY) { + scene_fire_common(hw, cookie, 0, offset, flags, NULL, 0, rca); + xpsb_kick_ta(hw); + return 0; + } + + /* Clean scene: bind the parameter buffer and start the tiler. */ + xpsb_wr(hw, SGX_TA_PARAM_BASE, offset + cookie[8]); + xpsb_wr(hw, SGX_DPM_ZLS_ENABLE, 0); + xpsb_wr(hw, SGX_DPM_REQUEST, 2); + xpsb_wr(hw, SGX_DPM_REQUEST2, 2); + xpsb_wr(hw, SGX_TA_CTRL2, 1); + xpsb_wrb(hw, SGX_TA_START, 0x80000000); + if (flags & PSB_SCENE_FLAG_SETUP) { + emit_scene_setup(hw, cookie, offset); + xpsb_wr(hw, SGX_TA_STATE, 0); + } + xpsb_kick_wait_clear(hw, 0, 0, 0, SGX_EVENT_STATUS, SGX_EV_TA_SETUP, + SGX_EV_TA_SETUP, SGX_EVENT_HOST_CLEAR, + SGX_EV_TA_SETUP); + xpsb_kick_ta(hw); + return 0; +} + +/* Xpsb_oom_abort at 0x3cf0: report the DPM status and select the recovery + * mode. It does not grow the parameter buffer; the reply asks the scheduler + * for a partial render (raster the scene, then resume the tiler). */ +int xpsb_oom_abort(const struct xpsb_hw *hw, uint32_t *cookie, + uint32_t *bca, uint32_t *rca, uint32_t *oflags) +{ + uint32_t status = xpsb_rd(hw, SGX_DPM_STATUS); + + /* XPSB_OOM_GLOBAL=1 forces the global abort the blob otherwise only + * escalates to. It flushes every macro tile instead of the one the + * DPM named, which is the difference between mask=0xffff and + * mask=1< + * + * SPDX-License-Identifier: MIT + */ +#ifndef _XPSB_HW_H_ +#define _XPSB_HW_H_ + +#include + +/* SGX registers touched by the responder. Names from psb_reg.h where the GPL + * kernel defines them; the rest keep their raw offsets. */ +#define SGX_CLKGATECTL 0x0000 +#define SGX_CORE_REVISION 0x0014 +#define SGX_EVENT_HOST_ENABLE2 0x0110 +#define SGX_EVENT_HOST_CLEAR2 0x0114 +#define SGX_EVENT_STATUS2 0x0118 +#define SGX_EVENT_STATUS 0x012c +#define SGX_EVENT_HOST_ENABLE 0x0130 +#define SGX_EVENT_HOST_CLEAR 0x0134 +#define SGX_EVENT_STATUS3 0x0138 +#define SGX_EVENT_HOST_CLEAR3 0x0140 + +#define SGX_TA_KICK 0x0200 +#define SGX_TA_MTILE_Y 0x020c +#define SGX_TA_MTILE_STRIDE 0x0214 +#define SGX_TA_MTILE_SIZE 0x0210 +#define SGX_TA_PIXEL_EXTENT 0x0248 +#define SGX_TA_REGION_BASE 0x021c +#define SGX_TA_PARAM_BASE 0x0220 +#define SGX_TA_MTILE_X 0x0208 +#define SGX_TA_ZLS_BASE 0x0234 +#define SGX_TA_CTRL2 0x024c +#define SGX_TA_START 0x0224 +#define SGX_TA_STATE 0x0274 + +#define SGX_3D_STATUS 0x0404 +#define SGX_3D_PARAM_BASE 0x0408 +#define SGX_3D_KICK 0x0428 +#define SGX_3D_CTRL 0x043c +#define SGX_ISP_CTRL 0x0414 +#define SGX_DPM_STATE 0x0480 +#define SGX_VISTEST_RESULT 0x0498 /* eight dwords */ +#define SGX_TA_SIG_BASE 0x0308 /* four dwords used by the watchdog */ +#define SGX_ISP_SIG_BASE 0x04cc /* six dwords used by the watchdog */ +#define SGX_MISC_SIG 0x0b08 + +#define SGX_DPM_RASTER_BASE 0x0600 +#define SGX_DPM_TA_BASE 0x0618 +#define SGX_DPM_HOSTA_BASE 0x0610 +#define SGX_DPM_HOSTD_BASE 0x0608 +#define SGX_DPM_PAGE_TABLE 0x062c +#define SGX_DPM_STATE_TABLE 0x0634 +#define SGX_DPM_ALLOC_MODE 0x0630 +#define SGX_DPM_TA_GLOBAL 0x0638 +#define SGX_DPM_TA_LOCAL_A 0x0620 +#define SGX_DPM_TA_LOCAL_B 0x0624 +#define SGX_DPM_TA_LOCAL_C 0x0628 +#define SGX_DPM_TA_TAIL 0x0668 +#define SGX_DPM_RASTER_GLOBAL 0x0660 +#define SGX_DPM_LOAD_TA 0x0684 +#define SGX_DPM_LOAD_RASTER 0x0680 +#define SGX_DPM_LOAD_HOSTA 0x0688 +#define SGX_DPM_LOAD_HOSTD 0x0690 +#define SGX_DPM_CONTEXT 0x063c +#define SGX_DPM_PARTIAL 0x0658 +#define SGX_DPM_ZLS_ENABLE 0x065c +#define SGX_DPM_REQUEST 0x0694 +#define SGX_DPM_REQUEST2 0x0698 +#define SGX_DPM_MODE 0x069c +#define SGX_DPM_STATUS 0x0720 +#define SGX_DPM_FREE 0x0724 +#define SGX_DPM_LOAD_KICK 0x06a8 + +#define SGX_USE_CTRL 0x0804 +#define SGX_TSP_CTRL 0x0a00 +#define SGX_PDS_CTRL 0x0a08 +#define SGX_MTE_CTRL 0x0a58 +#define SGX_DPM_LIMIT_A 0x0a74 +#define SGX_DPM_LIMIT_B 0x0a78 +#define SGX_DPM_LIMIT_C 0x0a7c +#define SGX_DPM_LIMIT_D 0x0a80 +#define SGX_DPM_THRESH_A 0x0aac +#define SGX_DPM_THRESH_B 0x0abc +#define SGX_TA_FLUSH 0x0ad4 +#define SGX_TA_FLUSH2 0x0ae0 +#define SGX_BIF_TA_FLUSH 0x0c90 +#define SGX_BIF_3D_FLUSH 0x0cb0 +#define SGX_BIF_CTRL2 0x0ca0 +#define SGX_TA_DPM_STAT_A 0x0aa4 +#define SGX_TA_DPM_STAT_B 0x0aa8 + +/* Event bits used as completion handshakes. */ +#define SGX_EV_SW_EVENT (1u << 14) /* raises the kernel's user IRQ */ +#define SGX_EV_TA_LOAD 0x00400000u +#define SGX_EV_DPM_A 0x00000010u +#define SGX_EV_DPM_B 0x00000020u +#define SGX_EV_DPM_C 0x00000040u +#define SGX_EV_DPM_D 0x00000200u +#define SGX_EV_TA_SETUP 0x00100a40u +#define SGX_EV_KICK_DONE 0x04000000u + +/* Per-screen SGX state owned by the responder. Field names replace the raw + * offsets of the binary's context; only the behaviour has to match. */ +struct xpsb_hw { + volatile uint8_t *reg; /* mapped MMIO, ctx+0x00 */ + uint32_t use_ctrl; /* value for SGX_USE_CTRL, ctx+0x0c */ + uint32_t vopt[42]; /* core-revision workarounds, ctx+0x14..0xb8 */ + uint32_t alloc_mode; /* mirror of SGX_DPM_ALLOC_MODE, ctx+0xc0 */ + /* GPU address and CPU mapping of one 0x88-byte pinned MMU buffer, the + * private DPM page-manager table the OOM resume binds. Xpsb.so keeps + * them at ctx+0xec (drmBO.offset) and ctx+0x138 (the map). */ + uint32_t oom_page_table;/* ctx+0xec */ + uint32_t (*regions)[2]; /* ctx+0x138, 17 8-byte macro-tile records */ + uint32_t last_sig; /* watchdog signature */ +}; + +/* The size of that buffer: 17 records of 8 bytes. Leaving oom_page_table zero + * points the DPM's page table at address zero, and the recovery ends in an MMU + * fault at 0x00000000 with no requestor bit - which is what it used to do. */ +#define XPSB_OOM_PT_BYTES 0x88 + +/* Indices into xpsb_hw.vopt, named for the three the responder consults. */ +#define VOPT_DPM_LIMIT 8 /* ctx+0x34: halves the DPM page limit */ +#define VOPT_LARGE_SCENE 18 /* ctx+0x5c: >2047 scenes keep 2x2 macro tiles */ +#define VOPT_WIDE_MTILE 20 /* ctx+0x64: 2 tiles per macro tile instead of 1/2 */ +#define VOPT_CLKGATE 38 /* ctx+0xac: extra USE clock-gate bit */ + +static inline uint32_t xpsb_rd(const struct xpsb_hw *hw, uint32_t off) +{ + return *(volatile uint32_t *)(hw->reg + off); +} + +static inline void xpsb_wr(const struct xpsb_hw *hw, uint32_t off, uint32_t v) +{ + *(volatile uint32_t *)(hw->reg + off) = v; +} + +/* Write then read back, which is how the binary orders every kick. */ +static inline void xpsb_wrb(const struct xpsb_hw *hw, uint32_t off, uint32_t v) +{ + xpsb_wr(hw, off, v); + (void)xpsb_rd(hw, off); +} + +void xpsb_set_vopt(struct xpsb_hw *hw); +void xpsb_sgx_initialize(struct xpsb_hw *hw); +void xpsb_sgx_uninit(struct xpsb_hw *hw); + +int xpsb_poll_reg(const struct xpsb_hw *hw, uint32_t off, uint32_t mask, + uint32_t val, uint32_t timeout_us); +int xpsb_kick_wait_clear(const struct xpsb_hw *hw, int do_kick, + uint32_t kick_off, uint32_t kick_val, + uint32_t stat_off, uint32_t mask, uint32_t val, + uint32_t clear_off, uint32_t clear_val); + +int xpsb_check_lockup(struct xpsb_hw *hw, uint32_t *value); + +int xpsb_scene_info(const struct xpsb_hw *hw, uint32_t w, uint32_t h, + uint32_t *cookie, uint32_t *size, + uint32_t *clear_p_start, uint32_t *clear_num_pages); +int xpsb_ta_mem_info(uint32_t pages, uint32_t *cookie, uint32_t *size); +int xpsb_ta_mem_load(struct xpsb_hw *hw, uint32_t pt_offset, + uint32_t param_offset, uint32_t flags, uint32_t *cookie); +int xpsb_scene_switch_fire(struct xpsb_hw *hw, uint32_t fire_flags, + uint32_t hw_context, uint32_t *cookie, + uint32_t offset, uint32_t engine, uint32_t flags, + const uint32_t *oom_cmds, uint32_t num_oom_cmds, + uint32_t *rca); +int xpsb_oom_abort(const struct xpsb_hw *hw, uint32_t *cookie, + uint32_t *bca, uint32_t *rca, uint32_t *oflags); + +int xpsb_kick_ta(const struct xpsb_hw *hw); +int xpsb_kick_render(const struct xpsb_hw *hw); + +#endif /* _XPSB_HW_H_ */ diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/xpsb_module.c mesa-26.2.2/src/gallium/drivers/sgx/xpsb_module.c --- mesa-26.2.2.orig/src/gallium/drivers/sgx/xpsb_module.c 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/xpsb_module.c 2026-09-08 10:57:36.684777161 +0200 @@ -0,0 +1,640 @@ +/* Open replacement for Xpsb.so - Xorg module and public API. + * + * xserver-xorg-video-psb loads this as a sub-module and resolves nine symbols + * from it (src/Xpsb.h in the DDX source is the contract). XpsbInit registers + * the XHW responder with the GPL kernel driver, without which no 3D + * submission is accepted at all; the composite and video entry points then + * execute the operation the DDX asked for. + * + * The surfaces the DDX passes are ordinary buffer objects, so an operation can + * be carried out either by building a frame for the SGX or through a CPU + * mapping. Each entry point offers it to xpsb_3d.c first and falls back to the + * mapping for anything that path declines, which keeps every entry point + * functional - in particular psbBlitYUV, whose result psbDisplayVideo ignores, + * so declining it would leave the video window blank rather than fall back. + * XPSB_NO_3D in the environment forces the CPU path everywhere. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#include +#include +#include +#include +#include + +#include "xf86Module.h" + +#include "Xpsb.h" + +#include "xpsb_3d.h" +#include "xpsb_comp.h" +#include "xpsb_shader.h" +#include "xpsb_xhw.h" +#include "xpsb_yuv.h" + +/* The DDX hands over the whole of BAR0; the SGX block sits 0x40000 into it, + * which is what psb_driver.h calls PSB_THALIA_OFFSET and what XpsbInit adds + * at 0x2c04 before storing the pointer its register accessors use. */ +#define XPSB_THALIA_OFFSET 0x00040000 + +#define XPSB_XHW_BO_SIZE 0x88 +#define XPSB_XHW_BO_FLAGS (DRM_BO_FLAG_READ | DRM_BO_FLAG_WRITE | \ + DRM_BO_FLAG_NO_EVICT | DRM_BO_FLAG_MEM_PRIV1) + +/* The out-of-memory resume binds CR_DPM_PAGE_TABLE to a private table rather + * than the scene's. XpsbInit allocates it at 0x2cf5 - 0x88 bytes of pinned + * MMU memory - and keeps its GPU address at ctx+0xec and its CPU mapping at + * ctx+0x138, which are xpsb_hw.oom_page_table and xpsb_hw.regions. */ +#define XPSB_DPM_BO_SIZE XPSB_OOM_PT_BYTES +#define XPSB_DPM_BO_FLAGS (DRM_BO_FLAG_READ | DRM_BO_FLAG_WRITE | \ + DRM_BO_FLAG_NO_EVICT | DRM_BO_FLAG_MEM_PRIV1) + +#define XPSB_MAX_MAPPED 4 + +struct xpsb_screen { + struct xpsb_xhw ctl; + drmBO bo; + drmBO dpm_bo; + pthread_t thread; + Bool running; + int exit_flag; + + struct xpsb_comp_state comp; + Bool comp_active; + + struct xpsb_3d g3d; + Bool use3d; /* the running composite is on the SGX */ + + drmBO *mapped[XPSB_MAX_MAPPED]; + int num_mapped; +}; + +static struct xpsb_screen *xpsb_screens[MAXSCREENS]; + +static struct xpsb_screen *xpsb_get(ScrnInfoPtr pScrn) +{ + if (pScrn->scrnIndex < 0 || pScrn->scrnIndex >= MAXSCREENS) + return NULL; + return xpsb_screens[pScrn->scrnIndex]; +} + +/* Mapping is refcounted in the kernel and also synchronises the buffer, which + * is exactly what libmm's mapBuf does for the DDX's own software paths. */ +static void *xpsb_map(struct xpsb_screen *s, drmBO *bo, unsigned int flags) +{ + void *virt; + + if (!bo || s->num_mapped >= XPSB_MAX_MAPPED) + return NULL; + if (drmBOMap(s->ctl.fd, bo, flags, 0, &virt)) + return NULL; + s->mapped[s->num_mapped++] = bo; + return virt; +} + +static void xpsb_unmap_all(struct xpsb_screen *s) +{ + while (s->num_mapped) + drmBOUnmap(s->ctl.fd, s->mapped[--s->num_mapped]); +} + +static int xpsb_filter(XpsbFilterFormats f) +{ + return (f == Xpsb_linear) ? XPSB_FILTER_LINEAR : XPSB_FILTER_NEAREST; +} + +static int surface_map(struct xpsb_screen *s, XpsbSurfacePtr srf, + struct xpsb_image *img, unsigned int flags) +{ + uint8_t *virt; + + if (xpsb_format_parse(srf->pictFormat, &img->fmt)) + return -1; + virt = xpsb_map(s, srf->buffer, flags); + if (!virt) + return -1; + + img->base = virt + srf->offset; + img->stride = (int)srf->stride; + img->w = srf->w; + img->h = srf->h; + img->filter = xpsb_filter(srf->magFilter); + img->umode = (int)srf->uMode; + img->vmode = (int)srf->vMode; + return 0; +} + +static void surface_3d(struct xpsb_3d_surf *o, XpsbSurfacePtr srf) +{ + memset(o, 0, sizeof *o); + if (!srf) + return; + o->handle = srf->buffer ? srf->buffer->handle : 0; + o->offset = srf->offset; + o->pict_format = srf->pictFormat; + o->w = srf->w; + o->h = srf->h; + o->stride = srf->stride; + o->umode = srf->uMode; + o->vmode = srf->vMode; + o->minfilter = (unsigned)xpsb_filter(srf->minFilter); + o->magfilter = (unsigned)xpsb_filter(srf->magFilter); +} + +_X_EXPORT Bool XpsbInit(ScrnInfoPtr pScrn, CARD8 *map, int drmFD) +{ + struct xpsb_screen *s; + struct drm_psb_xhw_init_arg init; + void *virt = NULL, *dpm_virt = NULL; + + if (pScrn->scrnIndex < 0 || pScrn->scrnIndex >= MAXSCREENS) + return FALSE; + if (xpsb_screens[pScrn->scrnIndex]) + return TRUE; + + s = calloc(1, sizeof *s); + if (!s) + return FALSE; + + s->ctl.fd = drmFD; + s->ctl.hw.reg = (volatile uint8_t *)map + XPSB_THALIA_OFFSET; + s->ctl.exit_flag = &s->exit_flag; + + if (drmBOCreate(drmFD, XPSB_XHW_BO_SIZE, 0, NULL, + XPSB_XHW_BO_FLAGS, 0, &s->bo)) { + xf86DrvMsg(pScrn->scrnIndex, X_ERROR, + "Xpsb: could not allocate the XHW communications " + "buffer: %s\n", strerror(errno)); + goto err_free; + } + if (drmBOMap(drmFD, &s->bo, DRM_BO_FLAG_READ | DRM_BO_FLAG_WRITE, + 0, &virt)) { + xf86DrvMsg(pScrn->scrnIndex, X_ERROR, + "Xpsb: could not map the XHW communications " + "buffer: %s\n", strerror(errno)); + goto err_bo; + } + s->ctl.shared = virt; + memset(virt, 0, XPSB_XHW_BO_SIZE); + + memset(&init, 0, sizeof init); + init.operation = PSB_XHW_INIT; + init.buffer_handle = s->bo.handle; + if (drmCommandWrite(drmFD, DRM_PSB_XHW_INIT, &init, sizeof init)) { + xf86DrvMsg(pScrn->scrnIndex, X_ERROR, + "Xpsb: the kernel rejected the XHW registration: " + "%s\n", strerror(errno)); + goto err_map; + } + + /* Without this the OOM resume would leave CR_DPM_PAGE_TABLE at zero + * and the DPM would fault at GPU address 0 as soon as the tiler + * resumes (Xpsb.so 0x47f3, 0x4883). */ + if (drmBOCreate(drmFD, XPSB_DPM_BO_SIZE, 0, NULL, XPSB_DPM_BO_FLAGS, + DRM_BO_HINT_DONT_FENCE, &s->dpm_bo)) { + xf86DrvMsg(pScrn->scrnIndex, X_ERROR, + "Xpsb: could not allocate the DPM page-manager " + "table: %s\n", strerror(errno)); + goto err_map; + } + if (drmBOMap(drmFD, &s->dpm_bo, DRM_BO_FLAG_READ | DRM_BO_FLAG_WRITE, + 0, &dpm_virt)) { + xf86DrvMsg(pScrn->scrnIndex, X_ERROR, + "Xpsb: could not map the DPM page-manager table: " + "%s\n", strerror(errno)); + goto err_dpm_bo; + } + memset(dpm_virt, 0, XPSB_DPM_BO_SIZE); + s->ctl.hw.regions = dpm_virt; + s->ctl.hw.oom_page_table = (uint32_t)s->dpm_bo.offset; + + xpsb_set_vopt(&s->ctl.hw); + xpsb_sgx_initialize(&s->ctl.hw); + + if (pthread_create(&s->thread, NULL, xpsb_xhw_thread, &s->ctl)) { + xf86DrvMsg(pScrn->scrnIndex, X_ERROR, + "Xpsb: could not start the XHW responder thread.\n"); + goto err_dpm_map; + } + s->running = TRUE; + xpsb_screens[pScrn->scrnIndex] = s; + + /* One frame per screen. xpsb_frame_init allocates nine buffer objects + * and sizes the render target for the whole screen, so it runs here + * and never per operation; failing it only costs the accelerated + * path, which is why it is not an error. */ + if (xpsb_3d_init(&s->g3d, drmFD, pScrn->virtualX, pScrn->virtualY)) + xf86DrvMsg(pScrn->scrnIndex, X_INFO, + "Xpsb: the 3D path is unavailable%s, operations " + "run on the CPU.\n", + xpsb_3d_caps() ? "" : " (XPSB_NO_3D)"); + + xf86DrvMsg(pScrn->scrnIndex, X_INFO, + "Xpsb: open XHW responder active, SGX core revision " + "%u.%u.%u.\n", + (xpsb_rd(&s->ctl.hw, SGX_CORE_REVISION) >> 16) & 0xff, + (xpsb_rd(&s->ctl.hw, SGX_CORE_REVISION) >> 8) & 0xff, + xpsb_rd(&s->ctl.hw, SGX_CORE_REVISION) & 0xff); + return TRUE; + +err_dpm_map: + drmBOUnmap(drmFD, &s->dpm_bo); +err_dpm_bo: + drmBOUnreference(drmFD, &s->dpm_bo); +err_map: + drmBOUnmap(drmFD, &s->bo); +err_bo: + drmBOUnreference(drmFD, &s->bo); +err_free: + free(s); + return FALSE; +} + +_X_EXPORT void XpsbTakeDown(ScrnInfoPtr pScrn) +{ + struct xpsb_screen *s = xpsb_get(pScrn); + struct drm_psb_xhw_init_arg init; + + if (!s) + return; + xpsb_screens[pScrn->scrnIndex] = NULL; + xpsb_unmap_all(s); + xpsb_3d_fini(&s->g3d); + + /* PSB_XHW_TAKEDOWN makes the kernel queue a TERMINATE request, which + * is what ends the responder loop; then the thread is joinable. */ + memset(&init, 0, sizeof init); + init.operation = PSB_XHW_TAKEDOWN; + init.buffer_handle = s->bo.handle; + drmCommandWrite(s->ctl.fd, DRM_PSB_XHW_INIT, &init, sizeof init); + + if (s->running) + pthread_join(s->thread, NULL); + + xpsb_sgx_uninit(&s->ctl.hw); + drmBOUnmap(s->ctl.fd, &s->dpm_bo); + drmBOUnreference(s->ctl.fd, &s->dpm_bo); + drmBOUnmap(s->ctl.fd, &s->bo); + drmBOUnreference(s->ctl.fd, &s->bo); + free(s); +} + +/* The 3D attempt. Returns 0 when the operation is on the SGX from here to + * psb3DCompositeFinish; anything else means the CPU path has to run it, and + * nothing has been mapped or submitted. */ +static int prepare_3d(struct xpsb_screen *s, XpsbSurfacePtr dst, + XpsbSurfacePtr opTextures[], int numOpTextures) +{ + struct xpsb_3d_comp r; + int tex = 0; + + memset(&r, 0, sizeof r); + r.op = s->comp.op; + r.scalar_src = s->comp.scalar_src; + r.scalar_mask = s->comp.scalar_mask; + r.scalar = s->comp.scalar; + surface_3d(&r.dst, dst); + + if (!r.scalar_src && tex < numOpTextures) { + surface_3d(&r.src, opTextures[tex++]); + r.have_src = 1; + } + if (!r.scalar_mask && tex < numOpTextures) { + surface_3d(&r.mask, opTextures[tex++]); + r.have_mask = 1; + } + return xpsb_3d_composite_begin(&s->g3d, &r); +} + +/* Returns non-zero when the operation is accepted; psbExaPrepareComposite3D + * turns a zero into an EXA software fallback. The choice between the SGX and + * the CPU is made here and is final, because psb3DCompositeQuad cannot report + * a failure. On the CPU path the surfaces stay mapped until + * psb3DCompositeFinish, so every quad in between is guaranteed to complete. */ +_X_EXPORT int psb3DPrepareComposite(ScrnInfoPtr pScrn, XpsbSurfacePtr dst, + XpsbSurfacePtr opTextures[], + int numOpTextures, int compOp, + unsigned int scalar, Bool scalarSrc, + Bool scalarMask) +{ + struct xpsb_screen *s = xpsb_get(pScrn); + int tex = 0; + + if (!s || compOp < 0 || compOp >= XPSB_NUM_OPS) + return 0; + if (s->comp_active) + xpsb_unmap_all(s); + if (s->use3d) + xpsb_3d_composite_finish(&s->g3d); + + memset(&s->comp, 0, sizeof s->comp); + s->comp.op = compOp; + s->comp.scalar_src = scalarSrc ? 1 : 0; + s->comp.scalar_mask = scalarMask ? 1 : 0; + s->comp.scalar = scalar; + s->comp_active = FALSE; + s->use3d = FALSE; + + if (!prepare_3d(s, dst, opTextures, numOpTextures)) { + s->use3d = TRUE; + return 1; + } + + if (surface_map(s, dst, &s->comp.dst, + DRM_BO_FLAG_READ | DRM_BO_FLAG_WRITE)) + goto fail; + + if (!s->comp.scalar_src) { + if (tex >= numOpTextures || + surface_map(s, opTextures[tex++], &s->comp.src, + DRM_BO_FLAG_READ)) + goto fail; + } + if (!s->comp.scalar_mask) { + if (tex >= numOpTextures || + surface_map(s, opTextures[tex++], &s->comp.mask, + DRM_BO_FLAG_READ)) + goto fail; + } + + s->comp_active = TRUE; + return 1; + +fail: + xpsb_unmap_all(s); + return 0; +} + +_X_EXPORT void psb3DCompositeQuad(ScrnInfoPtr pScrn, float vertices[]) +{ + struct xpsb_screen *s = xpsb_get(pScrn); + struct xpsb_3d_quad q3; + struct xpsb_quad q; + const float *v0, *v3; + int stride, coord = XPSB_VOFFSET_UT0; + + if (!s || (!s->comp_active && !s->use3d)) + return; + + /* psbExaComposite3D packs four floats per vertex when either source + * or mask is scalar, six when both are textures. */ + stride = (s->comp.scalar_src || s->comp.scalar_mask) ? 4 : 6; + v0 = &vertices[0]; + v3 = &vertices[3 * stride]; + + memset(&q, 0, sizeof q); + q.x0 = (int)v0[XPSB_VOFFSET_X]; + q.y0 = (int)v0[XPSB_VOFFSET_Y]; + q.x1 = (int)v3[XPSB_VOFFSET_X]; + q.y1 = (int)v3[XPSB_VOFFSET_Y]; + + if (!s->comp.scalar_src) { + q.su0 = v0[coord]; + q.sv0 = v0[coord + 1]; + q.su1 = v3[coord]; + q.sv1 = v3[coord + 1]; + coord += 2; + } + if (!s->comp.scalar_mask) { + q.mu0 = v0[coord]; + q.mv0 = v0[coord + 1]; + q.mu1 = v3[coord]; + q.mv1 = v3[coord + 1]; + } + + if (!s->use3d) { + xpsb_comp_quad(&s->comp, &q); + return; + } + + q3.x0 = (float)q.x0; + q3.y0 = (float)q.y0; + q3.x1 = (float)q.x1; + q3.y1 = (float)q.y1; + if (!s->comp.scalar_src) { + q3.u0 = q.su0; q3.v0 = q.sv0; + q3.u1 = q.su1; q3.v1 = q.sv1; + } else { + q3.u0 = q.mu0; q3.v0 = q.mv0; + q3.u1 = q.mu1; q3.v1 = q.mv1; + } + /* Nothing to report and nothing to retreat to: the surfaces were never + * mapped for the CPU path and earlier quads may already be on the + * panel, so a failure here leaves the operation incomplete. That is + * why the decision had to be final in psb3DPrepareComposite. */ + xpsb_3d_composite_quad(&s->g3d, &q3); +} + +/* Opposite convention to psb3DPrepareComposite: the closed module returns + * XpsbFlush3D's result verbatim, so zero means the flush succeeded. */ +_X_EXPORT int psb3DCompositeFinish(ScrnInfoPtr pScrn) +{ + struct xpsb_screen *s = xpsb_get(pScrn); + + if (!s) + return -1; + if (s->use3d) { + int ret = xpsb_3d_composite_finish(&s->g3d); + + s->use3d = FALSE; + return ret; + } + xpsb_unmap_all(s); + s->comp_active = FALSE; + return 0; +} + +/* The 3D attempt, on the same surfaces and texture coordinates the CPU blit + * below uses. Returns 0 when the frame rendered it, non-zero to fall back. */ +static int blit_yuv_3d(struct xpsb_screen *s, XpsbSurfacePtr dst, + XpsbSurfacePtr backTextures[], int numBackTextures, + unsigned int planarID, float texCoord0[], + float texCoord1[], float texCoord2[], + float conversion_data[]) +{ + struct xpsb_3d_video r; + int nplanes = xpsb_yuv_planes(planarID), i; + + if (nplanes == 0 || numBackTextures < nplanes) + return -1; + + memset(&r, 0, sizeof r); + r.fourcc = planarID; + r.nplanes = (unsigned)nplanes; + r.x = dst->x; + r.y = dst->y; + surface_3d(&r.dst, dst); + for (i = 0; i < nplanes; i++) + surface_3d(&r.plane[i], backTextures[i]); + + /* The same correction the CPU path makes: psbCopyPlanarNV12Data writes + * the interleaved chroma plane at the luma pitch. Describing the + * texture at the pitch the data is actually at is what lets the stride + * rule decline it rather than sample it wrongly. */ + if (planarID == XPSB_FOURCC_NV12) + r.plane[1].stride = r.plane[0].stride; + + /* psbBlitYUV overloads the last argument: XPSB_3D_CONV_N floats for a + * packed FOURCC, psb_pack_coeffs' nine fixed-point words for a planar + * one. Leaving conv NULL for the planar case declines rather than feeds + * the shader integers as floats. */ + if (nplanes == 1) + r.conv = conversion_data; + + r.u0 = texCoord0[0]; + r.v0 = texCoord0[1]; + r.u1 = texCoord1[0]; + r.v1 = texCoord2[1]; + return xpsb_3d_video_blit(&s->g3d, &r); +} + +static int blit_yuv(struct xpsb_screen *s, XpsbSurfacePtr dst, + XpsbSurfacePtr backTextures[], int numBackTextures, + unsigned int planarID, float texCoord0[], + float texCoord1[], float texCoord2[], + float conversion_data[]) +{ + struct xpsb_image dimg; + struct xpsb_yuv_src src; + struct xpsb_yuv_conv conv; + int nplanes = xpsb_yuv_planes(planarID); + int i, ret = 0; + uint8_t *virt; + + if (nplanes == 0 || numBackTextures < nplanes) + return 0; + if (xpsb_format_parse(dst->pictFormat, &dimg.fmt)) + return 0; + + memset(&src, 0, sizeof src); + src.fourcc = planarID; + src.w = backTextures[0]->w; + src.h = backTextures[0]->h; + src.filter = xpsb_filter(backTextures[0]->magFilter); + + for (i = 0; i < nplanes; i++) { + virt = xpsb_map(s, backTextures[i]->buffer, DRM_BO_FLAG_READ); + if (!virt) + goto out; + src.plane[i] = virt + backTextures[i]->offset; + src.stride[i] = (int)backTextures[i]->stride; + } + + /* psbCopyPlanarNV12Data writes the interleaved chroma plane at the + * luma pitch, which is not what the surface's own stride says. */ + if (planarID == XPSB_FOURCC_NV12) + src.stride[1] = src.stride[0]; + + if (!conversion_data) + xpsb_yuv_conv_default(&conv); + else if (nplanes == 1) + xpsb_yuv_conv_packed(conversion_data, &conv); + else + xpsb_yuv_conv_planar((const uint32_t *)conversion_data, &conv); + + virt = xpsb_map(s, dst->buffer, + DRM_BO_FLAG_READ | DRM_BO_FLAG_WRITE); + if (!virt) + goto out; + + /* Bias the destination to the box origin so the blit works in box + * coordinates and cannot address outside it. */ + dimg.stride = (int)dst->stride; + dimg.base = virt + dst->offset + (ptrdiff_t)dst->y * dimg.stride + + (ptrdiff_t)dst->x * (dimg.fmt.bpp >> 3); + dimg.w = dst->w; + dimg.h = dst->h; + dimg.filter = XPSB_FILTER_NEAREST; + dimg.umode = dimg.vmode = XPSB_ADDR_CLAMP; + + xpsb_yuv_blit(&dimg, 0, 0, (int)dst->w, (int)dst->h, + texCoord0[0], texCoord0[1], texCoord1[0], texCoord2[1], + &src, &conv); + ret = 1; + +out: + xpsb_unmap_all(s); + return ret; +} + +_X_EXPORT int psbBlitYUV(ScrnInfoPtr pScrn, XpsbSurfacePtr dst, + XpsbSurfacePtr backTextures[], int numBackTextures, + Bool isPlanar, unsigned int planarID, + float texCoord0[], float texCoord1[], + float texCoord2[], int numCoord, + float conversion_data[]) +{ + struct xpsb_screen *s = xpsb_get(pScrn); + + (void)isPlanar; (void)numCoord; + if (!s) + return 0; + if (!blit_yuv_3d(s, dst, backTextures, numBackTextures, planarID, + texCoord0, texCoord1, texCoord2, conversion_data)) + return 1; + return blit_yuv(s, dst, backTextures, numBackTextures, planarID, + texCoord0, texCoord1, texCoord2, conversion_data); +} + +_X_EXPORT int psbBlitYUVDetear(ScrnInfoPtr pScrn, XpsbSurfacePtr dst, + XpsbSurfacePtr backTextures[], + int numBackTextures, Bool isPlanar, + unsigned int planarID, float texCoord0[], + float texCoord1[], float texCoord2[], + int numCoord, float conversion_data[]) +{ + struct xpsb_screen *s = xpsb_get(pScrn); + drmVBlank vbl; + + (void)isPlanar; (void)numCoord; + if (!s) + return 0; + + /* The closed module renders to a scratch surface and hands the kernel + * a delayed 2D blit that the display interrupt replays. Retracing + * before a synchronous copy reaches the same point without a + * submission, and blocks the caller for the same span. */ + memset(&vbl, 0, sizeof vbl); + vbl.request.type = DRM_VBLANK_RELATIVE; + vbl.request.sequence = 1; + drmWaitVBlank(s->ctl.fd, &vbl); + + if (!blit_yuv_3d(s, dst, backTextures, numBackTextures, planarID, + texCoord0, texCoord1, texCoord2, conversion_data)) + return 1; + return blit_yuv(s, dst, backTextures, numBackTextures, planarID, + texCoord0, texCoord1, texCoord2, conversion_data); +} + +/* The closed module does not define this at all - the DDX declares it in + * Xpsb.h and never calls it. Exporting a no-op keeps the header honest and + * cannot change behaviour, since the blit is synchronous and nothing is ever + * left outstanding to cancel. */ +_X_EXPORT int XpsbCmdCancelBlit(ScrnInfoPtr pScrn) +{ + (void)pScrn; + return 1; +} + +static XF86ModuleVersionInfo XpsbVersRec = { + "Xpsb", + MODULEVENDORSTRING, + MODINFOSTRING1, + MODINFOSTRING2, + XORG_VERSION_CURRENT, + 0, 36, 0, + ABI_CLASS_VIDEODRV, + ABI_VIDEODRV_VERSION, + MOD_CLASS_NONE, + {0, 0, 0, 0} +}; + +static pointer XpsbSetup(pointer module, pointer opts, int *errmaj, int *errmin) +{ + (void)module; (void)opts; (void)errmaj; (void)errmin; + return (pointer)1; +} + +_X_EXPORT XF86ModuleData XpsbModuleData = { &XpsbVersRec, XpsbSetup, NULL }; diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/xpsb_pds.c mesa-26.2.2/src/gallium/drivers/sgx/xpsb_pds.c --- mesa-26.2.2.orig/src/gallium/drivers/sgx/xpsb_pds.c 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/xpsb_pds.c 2026-09-08 10:57:36.684786881 +0200 @@ -0,0 +1,240 @@ +/* Open replacement for Xpsb.so - primary (texture-sampling) PDS programs. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + * + * The program a fragment draw needs on the pixel side: launch the USE task, + * then for each sampled unit issue one iterator (DOUTI) and one texture sample + * (DOUTT). Field positions are those of tools/isa-pds (README section 2, "Word + * layout" and "Opcode table"), which decodes the whole SGX535 capture corpus. + * + * Both shapes this file emits exist in that corpus verbatim: one unit as + * P_mt2, two units as P_mt3. Every word below is checked against them by + * test_pds.c, so nothing here is inferred from a shape that was never seen. + * + * Data segment layout, taken from those two captures: + * + * ds0[0] USE task control word 0 ds1[0] USE task control word 2 + * ds0[1] USE task control word 1 ds1[1+2t] unit t iterator control + * ds0[2+2t] unit t texture state 0 ds1[2+2t] unit t texture address + * ds0[3+2t] unit t texture state 1 + * + * The closed Xpsb.so swaps the two ds1 slots - address at ds1[1+2t], iterator + * at ds1[2+2t] (disasm/emit-pixel-shader.md section 2). Both are self + * consistent because each driver's DOUTT selects the matching slot; this file + * follows the capture, which is what the existing stream replays and what the + * relocations at heap+0x368 / +0x370 already target. + * + * Disassembly of the generated ntex == 2 program, so the bytes are auditable + * (tools/isa-pds/pds-dis -b 0x20010340 -d 16), byte-identical to P_mt3: + * + * DATA @20010340 16 dwords + * 20010340: 0018102c 00000000 03fe0090 6c01f01f 03fe0090 6c00f00f 00000000 00000000 + * 20010360: 00000020 0fc0a201 804ee000 0c00fa02 804ef000 00000000 00000000 00000000 + * CODE @20010380 6 dwords + * 20010380: 07000345 movs doutu, ds0[0], ds0[1], ds1[0], ds1[0] + * ; doutu: exe=+0x100 coff=0x10000 iterdep texdep | mode=parallel | trc=0 sdsoft + * 20010384: 070418a2 movs douti, ds1[1], ds0[2], ds0[3], ds0[3] + * ; douti: texissue=1 useissue=10 uselast dim=4D + * 20010388: 07042364 movs doutt, ds0[2], ds0[3], ds1[2], ds1[3] + * ; doutt: texture state {03fe0090 6c01f01f 804ee000} tex@0x804ee000 + * 2001038c: 070438a2 movs douti, ds1[3], ds0[2], ds0[3], ds0[3] + * ; douti: texissue=2 useissue=15 texlast dim=1D + * 20010390: 07084364 movs doutt, ds0[4], ds0[5], ds1[4], ds1[5] + * ; doutt: texture state {03fe0090 6c00f00f 804ef000} tex@0x804ef000 + * 20010394: af000000 halt + * + * The ntex == 1 program is the same minus the second DOUTI/DOUTT pair, and is + * byte-identical to what the working stream carries at heap+0x340. + */ +#include "xpsb_pds.h" + +#include +#include +#include + +/* The MOVS encoder and its operand names now live in xpsb_pds.h. */ +#define MOVS XPSB_MOVS + +#define S1L XPSB_PDS_S1L +#define S1H XPSB_PDS_S1H +#define S2L XPSB_PDS_S2L +#define S2H XPSB_PDS_S2H + +#define DOUTI XPSB_PDS_PORT_DOUTI +#define DOUTT XPSB_PDS_PORT_DOUTT +#define DOUTU XPSB_PDS_PORT_DOUTU + +#define HALT XPSB_PDS_HALTW + +static void put(uint32_t *data, unsigned *n, unsigned bank, unsigned idx, + uint32_t v) +{ + unsigned d = xpsb_pds_ds_dword(bank, idx); + + data[d] = v; + if (d + 1 > *n) + *n = d + 1; +} + +unsigned xpsb_pds_issue_regs(unsigned nissue, const struct xpsb_pds_issue *issue) +{ + unsigned i, n = 0; + + for (i = 0; i < nissue; i++) { + if (issue[i].useissue != XPSB_DOUTI_NONE) + n += issue[i].unpacked ? + xpsb_douti_regs(issue[i].usedim, + issue[i].usef16) : 1u; + if (issue[i].texissue != XPSB_DOUTI_NONE) + n++; + } + return n; +} + +int xpsb_pds_gen_primary_issues(unsigned nissue, + const struct xpsb_pds_issue *issue, + const uint32_t use[3], + uint32_t *data, unsigned *ndata, + uint32_t *code, unsigned *ncode, + unsigned *code_off) +{ + return xpsb_pds_gen_primary_issues_max(nissue, issue, use, data, ndata, + code, ncode, code_off, + XPSB_PDS_CAP_DATA); +} + +int xpsb_pds_gen_primary_issues_max(unsigned nissue, + const struct xpsb_pds_issue *issue, + const uint32_t use[3], + uint32_t *data, unsigned *ndata, + uint32_t *code, unsigned *ncode, + unsigned *code_off, unsigned max_data) +{ + unsigned n = 0, c = 0, i, t = 0, ntex = 0; + int lastuse = -1, lasttex = -1; + + if (!nissue || nissue > XPSB_PDS_MAX_ISSUE || !issue || !use) { + errno = EINVAL; + return -1; + } + for (i = 0; i < nissue; i++) { + if (issue[i].useissue != XPSB_DOUTI_NONE) + lastuse = (int)i; + if (issue[i].texissue != XPSB_DOUTI_NONE) { + lasttex = (int)i; + ntex++; + } + } + /* Sized before anything is written: the last issue's address slot is + * the highest data dword and the code is one word an issue, one a + * sampled unit, the launch and the halt. Four issues reach ds1[8], + * memory dword 24, past the caller's sixteen. */ + { + /* Sized by the slots each issue really reaches, through the + * one definition of that in xpsb_pds.h. A DOUTI reads only its + * own word; the address and state slots belong to the DOUTT, + * so only a texture issue needs them - reserving them for every + * issue put the ceiling at three when the part takes four. */ + unsigned hi = xpsb_pds_list_high_dw(nissue, 0); + + if (lasttex >= 0) { + unsigned t = xpsb_pds_addr_dw((unsigned)lasttex); + unsigned u = xpsb_pds_fmt_dw((unsigned)lasttex); + + if (t > hi) + hi = t; + if (u > hi) + hi = u; + } + if (hi >= max_data || hi >= XPSB_PDS_MAX_DATA || + 2 + nissue + ntex > XPSB_PDS_MAX_CODE) { + errno = ENOSPC; + return -1; + } + } + + memset(data, 0, max_data * sizeof *data); + put(data, &n, 0, 0, use[0]); + put(data, &n, 0, 1, use[1]); + put(data, &n, 1, 0, use[2]); + code[c++] = MOVS(0, 0, S1L, S1H, S2L, S2L, DOUTU); + + /* ds1[1 + 2i] is the issue's control word and ds1[2 + 2i] the texture + * address, with the state pair in ds0 - the layout the captured + * program uses, extended one slot pair per issue. */ + for (i = 0; i < nissue; i++) { + uint32_t w = issue[i].word ? issue[i].word : + xpsb_pds_douti(issue[i].texissue, issue[i].texdim, + issue[i].useissue, issue[i].usedim, + (issue[i].unpacked ? + XPSB_DOUTI_UNPACKED : 0u) | + (issue[i].unpacked && + issue[i].usef16 ? + XPSB_DOUTI_USEFORMAT_F16 : 0u) | + ((int)i == lasttex ? + XPSB_DOUTI_TEXLAST : 0u) | + ((int)i == lastuse ? + XPSB_DOUTI_USELAST : 0u)); + + if (issue[i].flatshade && !issue[i].word) + w = (w & ~XPSB_DOUTI_FLAT_MASK) | + ((uint32_t)(issue[i].flatshade - 1u) << + XPSB_DOUTI_FLAT_SHIFT); + put(data, &n, 1, 1 + 2 * i, w); /* xpsb_pds_douti_dw(i) */ + /* Only a texture issue's DOUTT reads these, and writing them + * for one that samples nothing grew the segment past what the + * region holds. */ + if (issue[i].texissue != XPSB_DOUTI_NONE) { + put(data, &n, 0, 2 + 2 * i, issue[i].ctl); + put(data, &n, 0, 3 + 2 * i, issue[i].fmt); + put(data, &n, 1, 2 + 2 * i, issue[i].addr); + } + code[c++] = MOVS(1, i, S2H, S1L, S1H, S1H, DOUTI); + if (issue[i].texissue != XPSB_DOUTI_NONE) { + code[c++] = MOVS(1 + i, 1 + i, S1L, S1H, S2L, S2H, + DOUTT); + t++; + } + } + (void)t; + code[c++] = HALT; + n = (n + 3) & ~3u; + + *ndata = n; + *ncode = c; + *code_off = n * 4; + return 0; +} + +int xpsb_pds_gen_primary(unsigned ntex, const struct xpsb_pds_tex *tex, + const uint32_t use[3], + uint32_t *data, unsigned *ndata, + uint32_t *code, unsigned *ncode, + unsigned *code_off) +{ + struct xpsb_pds_issue issue[XPSB_PDS_NTEX_MAX]; + unsigned t; + + /* The captured shape as an issue list: every unit samples its own + * coordinate set, and the iterated colour rides on the first one's + * issue. The words this produces are the captured words. */ + if (ntex < 1 || ntex > XPSB_PDS_NTEX_MAX || !tex || !use) { + errno = EINVAL; + return -1; + } + memset(issue, 0, sizeof issue); + for (t = 0; t < ntex; t++) { + issue[t].texissue = (uint8_t)t; + issue[t].texdim = 3; + issue[t].useissue = t ? XPSB_DOUTI_NONE : XPSB_DOUTI_COLOUR0; + issue[t].usedim = t ? 1 : 4; + issue[t].word = tex[t].itr; + issue[t].ctl = tex[t].ctl; + issue[t].fmt = tex[t].fmt; + issue[t].addr = tex[t].addr; + } + return xpsb_pds_gen_primary_issues(ntex, issue, use, data, ndata, + code, ncode, code_off); +} diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/xpsb_pds.h mesa-26.2.2/src/gallium/drivers/sgx/xpsb_pds.h --- mesa-26.2.2.orig/src/gallium/drivers/sgx/xpsb_pds.h 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/xpsb_pds.h 2026-09-08 10:57:36.684796732 +0200 @@ -0,0 +1,336 @@ +/* Open replacement for Xpsb.so - primary (texture-sampling) PDS programs. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#ifndef _XPSB_PDS_H_ +#define _XPSB_PDS_H_ + +#include + +#define XPSB_PDS_NTEX_MAX 2 +/* Issues one primary program may carry. Each costs two ds1 and two ds0 slots, + * and the code segment sits between the data segment and the vertex PDS + * program's own data at heap+0x3a0 - which is what caps this at three, not + * anything in the hardware. A wider list needs the primary program moved the + * way the driver already moves its per-record copies. */ +/* One per coordinate set, plus the dummy texture issue that keeps a + * textureless pixel task at full rate. */ +#define XPSB_PDS_MAX_ISSUE 8 +#define XPSB_PDS_MAX_DATA 48 +#define XPSB_PDS_MAX_CODE 24 /* 2 + one per issue + one per sampled unit, and the test hook */ +/* Data dwords the captured slot holds. The arrays above are sized for the + * widest list the spare block takes; this is what a program built where the + * capture had it may use, and it is the default budget so that a frame which + * fits there is laid out exactly as before. */ +#define XPSB_PDS_CAP_DATA 16 + +/* MOVS: [31:30] group 0, [29:27] type 0, [26:24] cc 7 (always), [23] SRC1SEL, + * [22:18] SRC1 (a ds0 pair), [17:13] SRC2 (a ds1 pair), [12:11] [10:9] [8:7] + * [6:5] the four swizzle slots, [4] unused on SGX535, [3:0] destination port. + * A slot reads dword 2*SRCn + (swiz & 1) of the bank its swizzle names. + * Shared with xpsb_heap.c, which emits the parameter heap's PDS programs from + * the same encoder rather than transcribing them. */ +/* The general form: [31:30] group, [29:27] type, [26:24] condition. Every + * MOVS-family word in the captured parameter heap re-encodes exactly through + * this - plain MOVS, MOVSA, and the conditional MOVS the vertex program uses. */ +#define XPSB_PDS_ENC(group, type, cc, src1, src2, s0, s1, s2, s3, dest) \ + (((uint32_t)(group) << 30) | ((uint32_t)(type) << 27) | \ + ((uint32_t)(cc) << 24) | ((uint32_t)(src1) << 18) | \ + ((uint32_t)(src2) << 13) | ((uint32_t)(s0) << 11) | \ + ((uint32_t)(s1) << 9) | ((uint32_t)(s2) << 7) | \ + ((uint32_t)(s3) << 5) | (uint32_t)(dest)) + +#define XPSB_PDS_CC_ALWAYS 7 +#define XPSB_PDS_CC_IF0 3 +#define XPSB_PDS_TYPE_MOVS 0 +#define XPSB_PDS_TYPE_MOVSA 5 + +#define XPSB_MOVS(src1, src2, s0, s1, s2, s3, dest) \ + XPSB_PDS_ENC(0, XPSB_PDS_TYPE_MOVS, XPSB_PDS_CC_ALWAYS, \ + src1, src2, s0, s1, s2, s3, dest) + +#define XPSB_PDS_S1L 0 +#define XPSB_PDS_S1H 1 +#define XPSB_PDS_S2L 2 +#define XPSB_PDS_S2H 3 + +#define XPSB_PDS_PORT_DOUTI 2 +#define XPSB_PDS_PORT_DOUTD 3 +#define XPSB_PDS_PORT_DOUTT 4 +#define XPSB_PDS_PORT_DOUTU 5 + +#define XPSB_PDS_HALTW 0xaf000000u + +/* One sampled unit. All four words are payloads of the TAG/USE interface, not + * of the PDS ISA: ctl/fmt/addr are the three dwords EURASIA_TAG_TEXTURE_STATE + * defines, itr is the DOUTI control word. The generator only places them. */ +struct xpsb_pds_tex { + uint32_t ctl; + uint32_t fmt; + uint32_t addr; + uint32_t itr; +}; + +/* DOUTI iterator control word. The fields are the vendor's + * psb_emit_ps_primary_compiled(): the low nibble is the coordinate set handed + * to the TAG and the [15:12] nibble is what is delivered to the USE, 15 in + * either meaning "none". usedim is how many primary-attribute registers the + * USE issue claims - one when the value is delivered packed, which is how the + * iterated colour arrives. + * + * A varying is texissue 15 with useissue set to its coordinate set; a sampled + * unit is the reverse; the vendor also merges the two into one word. */ +/* The field names are the DDK's (EURASIA_PDS_DOUTI_*), matched against the + * dropped gfx_linux_ddk hwdefs. Two long-standing misreadings corrected by + * them: bits [9:8] are not a coordinate width but TEXPROJ, the TAG's + * projection - 0 none, 1 divide by the RHW plane, 2 divide by the + * coordinate's own last component, 3 by its third - so the "texdim 3, always" + * this driver emits is TEXPROJ 2, the divide the perspective correction rides + * on; and bit 24 is not a use-enable but USEPERSPECTIVE, correcting the USE + * issue against the RHW plane the record's fourth position float carries. + * "texdim 2 stalls the render" was TEXPROJ_RHW waiting on a plane the frame + * then never declared. */ +#define XPSB_DOUTI_NONE 15u +#define XPSB_DOUTI_COLOUR0 10u /* the packed diffuse colour */ +#define XPSB_DOUTI_COLOUR1 11u +#define XPSB_DOUTI_TEXLAST (1u << 11) +#define XPSB_DOUTI_USEEN (1u << 24) /* USEPERSPECTIVE */ +#define XPSB_DOUTI_USELAST (1u << 25) +#define XPSB_DOUTI_UNPACKED (1u << 20) /* USECOLFLOAT */ +/* [27:26] is EURASIA_PDS_DOUTI_FLATSHADE (sgxdefs.h:3750-3755, the + * !defined(SGX545) branch this core takes): 0..2 take the value from vertex + * 0, 1 or 2 of the triangle and 3 interpolates it. Note which end is which - + * an all-zero field is flat from vertex 0, not gouraud, so GOURAUD is a value + * that has to be written, which is why XPSB_DOUTI_FMT_DEFAULT carries it. + * + * The field belongs to the USE half of the word. The TAG fields are [11:0] + * and every USE field is above them, this one among them; and where the + * vendor's own generator folds a sampling iterator into the issue before it, + * it copies EURASIA_DOUTI_TAG_MASK only - TEXISSUE, TEXCENTROID, TEXWRAP, + * TEXPROJ and TEXLASTISSUE, not FLATSHADE (codegen/pds/pds.c:152-157, + * 1176-1177). So a merged word keeps the corner its USE issue asked for, and + * a word that both samples and iterates may be flat: the vendor produces + * exactly that for a flat-shaded textured draw. What the hardware does not + * support is the other direction - the TAG cannot sample a coordinate set the + * program wants flat, "the hardware doesn't support non-dependent texture + * samples with flat shaded texture coordinates" + * (tools/intern/usc2/regpack.c:5722-5726, icvt_core.c:6112-6115) - and a + * texture coordinate delivered to the USE can be flat like any other varying + * (usc.h:1053-1055). + * + * The vendor gives an iterator VTX2 when the MTE's shade model is + * EURASIA_MTE_SHADE_VERTEX2 and the issue does not sample, gouraud otherwise: + * "colours are affected by colour-interpolation, texcoords are always gouraud + * shaded", opengles1/usegles.c:2240-2252. That test is on the sampler, not on + * the slot, so it covers a colour routed onto a coordinate set under + * FIX_HW_BRN_25211 as well as one on USEISSUE_V0. */ +#define XPSB_DOUTI_FLAT_SHIFT 26 +#define XPSB_DOUTI_FLAT_MASK (3u << XPSB_DOUTI_FLAT_SHIFT) +#define XPSB_DOUTI_FMT_DEFAULT (3u << 26) /* FLATSHADE_GOURAUD */ + +/* DOUTI[29:28], EURASIA_PDS_DOUTI_USEFORMAT (sgxdefs.h:3757-3763). The field + * exists on this core: it is guarded by SGX_FEATURE_EXTENDED_USE_ALU, which + * SGX535 defines (sgxfeaturedefs.h:215). Code 0 is INT8, which the vendor + * documents as "backwards compatible with SGX535" - INT8 only for a base or + * highlight colour with USECOLFLOAT clear, F32 in every other case + * (opengles2/use.c:2185-2192) - so it is what every frame here has been + * emitting for a coordinate set. Code 2 delivers the iterated value as half + * floats instead, two components to a primary attribute register. */ +#define XPSB_DOUTI_USEFORMAT_SHIFT 28 +#define XPSB_DOUTI_USEFORMAT_INT8 (0u << XPSB_DOUTI_USEFORMAT_SHIFT) +#define XPSB_DOUTI_USEFORMAT_INT10 (1u << XPSB_DOUTI_USEFORMAT_SHIFT) +#define XPSB_DOUTI_USEFORMAT_F16 (2u << XPSB_DOUTI_USEFORMAT_SHIFT) + +/* Primary attribute registers an iterated value of ncomp components claims. + * The USEDIM field always counts components - the vendor writes uCoordDim - 1 + * into it whatever the format (opengles2/use.c:2179), and its own dummy + * iteration pairs a four-component F16 load with a two-register data size + * (usp/finalise.c:3646-3652) - so the register count follows from the format: + * one component to a register as F32, two as F16 (usp_inputdata.c:1401-1409, + * uCompsPerReg 2 and uRegChansPerComp 2). */ +static inline unsigned xpsb_douti_regs(unsigned ncomp, int f16) +{ + unsigned n = ncomp ? ncomp : 1u; + + return f16 ? (n + 1u) / 2u : n; +} + +/* DOUTU word 0's dependency bits: the task waits for the issues it is told to + * expect, so they have to match the DOUTI/DOUTT list exactly. */ +/* DOUTU word 1's punch-through field, EURASIA_PDS_DOUTU1_PUNCHTHROUGH: 0 is + * disabled, 1 is phase one, 2 phase two. The vendor sets phase one whenever + * the program can discard. */ +#define XPSB_DOUTU1_PUNCHTHROUGH (3u << 24) +#define XPSB_DOUTU1_PUNCHTHROUGH_PHASE1 (1u << 24) + +#define XPSB_DOUTU_ITERDEP (1u << 19) +#define XPSB_DOUTU_TEXDEP (1u << 20) + +static inline uint32_t xpsb_pds_douti(unsigned texissue, unsigned texdim, + unsigned useissue, unsigned usedim, + uint32_t flags) +{ + uint32_t w = (texissue & 0xfu) | ((useissue & 0xfu) << 12) | + XPSB_DOUTI_FMT_DEFAULT | flags; + + + if (texissue != XPSB_DOUTI_NONE && texdim) + w |= ((texdim - 1u) & 3u) << 8; + if (useissue != XPSB_DOUTI_NONE) + w |= (((usedim ? usedim : 1u) - 1u) & 3u) << 22 | + XPSB_DOUTI_USEEN; + return w; +} + +/* One issue of the primary program: what it hands the TAG, what it hands the + * USE, and how many registers the latter claims. */ +struct xpsb_pds_issue { + uint8_t texissue; /* coordinate set to sample, or 15 */ + uint8_t texdim; /* 2 or 3; only read when sampling */ + uint8_t useissue; /* what to iterate, or 15 */ + uint8_t usedim; /* components iterated */ + uint8_t unpacked; /* delivered as floats rather than packed */ + /* Which coordinate set this issue iterates, or 0xff. The useissue + * field cannot say: a set carried on a colour iterator names V0 or + * V1 there, and the caller's base array is indexed by set. */ + uint8_t useset; + /* Zero interpolates; 1..3 take the value from vertex 0, 1 or 2 - the + * hardware field offset by one, so it shares an encoding with the + * MTE's shade model. Only the USE side of the issue is affected; what + * the issue samples is iterated gouraud whatever this says. */ + uint8_t flatshade; + /* Delivered as half floats: usedim still counts components, but two + * of them share a register, so the issue claims half as many. Only + * read for an unpacked issue - the packed colour is one dword and the + * field is not honoured for it. */ + uint8_t usef16; + /* A texture issue added only so the frame has one - a pixel task with + * no TAG issue runs an order of magnitude slower on this part, and the + * vendor driver never submits one - so it is the first thing dropped + * when the program does not fit. */ + uint8_t dummy; + uint32_t word; /* control word verbatim; 0 = build one */ + uint32_t ctl, fmt, addr; /* texture state, when sampling */ +}; + +/* As xpsb_pds_gen_primary(), but from an explicit issue list: one DOUTI per + * issue and a DOUTT only for those that sample. The last-issue flags are + * applied to the last of each kind. */ +/* As xpsb_pds_gen_primary_issues(), with the data budget named: the region the + * program is being built in bounds the list, not the caller's array. */ +int xpsb_pds_gen_primary_issues_max(unsigned nissue, + const struct xpsb_pds_issue *issue, + const uint32_t use[3], + uint32_t *data, unsigned *ndata, + uint32_t *code, unsigned *ncode, + unsigned *code_off, unsigned max_data); +int xpsb_pds_gen_primary_issues(unsigned nissue, + const struct xpsb_pds_issue *issue, + const uint32_t use[3], + uint32_t *data, unsigned *ndata, + uint32_t *code, unsigned *ncode, + unsigned *code_off); + +/* How many primary-attribute registers an issue list claims: a packed value + * occupies one register whatever its usedim says, an unpacked one as many as + * xpsb_douti_regs() gives for its format, and every sampled unit's texel + * occupies one more. */ +unsigned xpsb_pds_issue_regs(unsigned nissue, const struct xpsb_pds_issue *issue); + +/* Build the primary PDS program for ntex sampled units. use[] is the 3-dword + * USE task control block. data and code must have room for XPSB_PDS_MAX_DATA + * and XPSB_PDS_MAX_CODE dwords; *code_off is the byte offset of the code + * segment from the start of the data segment. Returns 0, or -1 with EINVAL. */ +int xpsb_pds_gen_primary(unsigned ntex, const struct xpsb_pds_tex *tex, + const uint32_t use[3], + uint32_t *data, unsigned *ndata, + uint32_t *code, unsigned *ncode, + unsigned *code_off); + +/* Memory dword backing ds[idx]: SGX535 has no PDS_DATA_INTERLEAVE_2DWORDS, + * so PDS_NUM_DWORDS_PER_ROW is 8 and the banks alternate every 8 dwords. */ +static inline unsigned xpsb_pds_ds_dword(unsigned bank, unsigned idx) +{ + return (idx / 8) * 16 + bank * 8 + idx % 8; +} + +/* ---- the primary PDS data segment's layout, in one place ---- + * + * Every slot an issue owns, named rather than stepped. Each of these was open + * coded at some call site once, with a stride that is right for the first + * three issues and wrong for the fourth - ds0[8] is memory dword 16, not 8 - + * and each wrong one showed up as a core stall with no MMU fault rather than + * as anything a reader could see. There is one definition now and the callers + * ask it. */ +static inline unsigned xpsb_pds_ctl_dw(unsigned issue) +{ + return xpsb_pds_ds_dword(0, 2 + 2 * issue); +} +static inline unsigned xpsb_pds_fmt_dw(unsigned issue) +{ + return xpsb_pds_ds_dword(0, 3 + 2 * issue); +} +static inline unsigned xpsb_pds_addr_dw(unsigned issue) +{ + return xpsb_pds_ds_dword(1, 2 + 2 * issue); +} +static inline unsigned xpsb_pds_douti_dw(unsigned issue) +{ + return xpsb_pds_ds_dword(1, 1 + 2 * issue); +} + +/* The highest data dword a list of n issues reaches. `samples` says whether + * the last issue's DOUTT slots count: an iterating issue reads only its DOUTI + * word, and reserving the texture slots for it put the ceiling at three when + * the part would take four. */ +static inline unsigned xpsb_pds_list_high_dw(unsigned n, int samples) +{ + unsigned hi, q; + + if (!n) + return 0; + hi = xpsb_pds_douti_dw(n - 1); + if (!samples) + return hi; + q = xpsb_pds_addr_dw(n - 1); + if (q > hi) + hi = q; + q = xpsb_pds_fmt_dw(n - 1); + if (q > hi) + hi = q; + return hi; +} + +/* Whether that list fits a segment of max_data dwords. One test, so the + * builder and everything that asks before building cannot disagree - they did, + * and the disagreement appended an issue that was then shed while the register + * bases went on counting it. */ +static inline int xpsb_pds_list_fits(unsigned n, int samples, unsigned max_data) +{ + return n && n <= XPSB_PDS_MAX_ISSUE && + xpsb_pds_list_high_dw(n, samples) < max_data; +} + +/* Issues the generator can actually build. The data segment lays issue i's + * texture address at ds1[2 + 2i], and ds1[8] is memory dword 24 - past the + * sixteen the heap's primary PDS region holds before the vertex descriptor at + * XPSB_PDS_VTXDESC_OFF - so a fourth issue overruns however it is filled. + * XPSB_PDS_MAX_ISSUE bounds the caller's array; this bounds the program. */ +static inline int xpsb_pds_issues_fit(unsigned n) +{ + /* What this bounds is the dummy TAG appended to a list, and that is a + * texture issue - so its DOUTT slots count, not the DOUTI alone. */ + return xpsb_pds_list_fits(n, 1, XPSB_PDS_CAP_DATA); +} + +/* PDS state descriptor word: size in [31:24], (addr - PSB_MEM_PDS_START) >> 4 + * in [23:0]. */ +static inline uint32_t xpsb_pds_desc_word(uint32_t gpu_addr, unsigned size) +{ + return ((uint32_t)size << 24) | + (((gpu_addr - 0x20000000u) >> 4) & 0x00ffffffu); +} + +#endif /* _XPSB_PDS_H_ */ diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/xpsb_pixel.c mesa-26.2.2/src/gallium/drivers/sgx/xpsb_pixel.c --- mesa-26.2.2.orig/src/gallium/drivers/sgx/xpsb_pixel.c 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/xpsb_pixel.c 2026-09-08 10:57:36.684804432 +0200 @@ -0,0 +1,199 @@ +/* Open replacement for Xpsb.so - PICT format handling and texture sampling. + * + * The unpack path reproduces psbPixelARGB8888 in the DDX bit for bit, + * including its bit-expansion rule, so a scalar pixel the DDX computed and a + * texel this file reads agree. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#include +#include + +#include "xpsb_pixel.h" + +#define PICT_FORMAT_BPP(f) (((f) >> 24) & 0xff) +#define PICT_FORMAT_TYPE(f) (((f) >> 16) & 0xff) +#define PICT_FORMAT_A(f) (((f) >> 12) & 0x0f) +#define PICT_FORMAT_R(f) (((f) >> 8) & 0x0f) +#define PICT_FORMAT_G(f) (((f) >> 4) & 0x0f) +#define PICT_FORMAT_B(f) (((f) ) & 0x0f) + +#define PICT_TYPE_A 1 +#define PICT_TYPE_ARGB 2 +#define PICT_TYPE_ABGR 3 + +#define IDX_A 0 +#define IDX_R 1 +#define IDX_G 2 +#define IDX_B 3 + +/* The DDX widens a component by shifting it up and filling the low bits from + * its least significant bit rather than by replication. */ +static unsigned int bit_expand(unsigned int v, unsigned int bits) +{ + unsigned int mask = (1u << (8 - bits)) - 1u; + unsigned int tmp = v << (8 - bits); + + return (v & 1) ? (tmp | mask) : tmp; +} + +int xpsb_format_parse(unsigned int pict_format, struct xpsb_format *f) +{ + unsigned int order[4], i, shift = 0; + + f->bpp = PICT_FORMAT_BPP(pict_format); + if (f->bpp != 8 && f->bpp != 16 && f->bpp != 32) + return -1; + + f->bits[IDX_A] = PICT_FORMAT_A(pict_format); + f->bits[IDX_R] = PICT_FORMAT_R(pict_format); + f->bits[IDX_G] = PICT_FORMAT_G(pict_format); + f->bits[IDX_B] = PICT_FORMAT_B(pict_format); + f->has_alpha = f->bits[IDX_A] != 0; + + switch (PICT_FORMAT_TYPE(pict_format)) { + case PICT_TYPE_A: + if (!f->bits[IDX_A]) + return -1; + f->shift[IDX_A] = 0; + f->shift[IDX_R] = f->shift[IDX_G] = f->shift[IDX_B] = 0; + return 0; + case PICT_TYPE_ARGB: + order[0] = IDX_B; order[1] = IDX_G; + order[2] = IDX_R; order[3] = IDX_A; + break; + case PICT_TYPE_ABGR: + order[0] = IDX_R; order[1] = IDX_G; + order[2] = IDX_B; order[3] = IDX_A; + break; + default: + return -1; + } + + for (i = 0; i < 4; i++) { + f->shift[order[i]] = shift; + shift += f->bits[order[i]]; + } + if (shift > f->bpp) + return -1; + return 0; +} + +static uint32_t raw_load(const struct xpsb_format *f, const uint8_t *p) +{ + if (f->bpp == 8) + return *p; + if (f->bpp == 16) + return *(const uint16_t *)p; + return *(const uint32_t *)p; +} + +static void raw_store(const struct xpsb_format *f, uint8_t *p, uint32_t v) +{ + if (f->bpp == 8) + *p = (uint8_t)v; + else if (f->bpp == 16) + *(uint16_t *)p = (uint16_t)v; + else + *(uint32_t *)p = v; +} + +uint32_t xpsb_pix_load(const struct xpsb_format *f, const uint8_t *p) +{ + uint32_t pixel = raw_load(f, p); + uint32_t argb = 0; + unsigned int i; + + for (i = 0; i < 4; i++) { + unsigned int bits = f->bits[i], v; + + if (!bits) { + /* No alpha channel means opaque; a colour channel a + * format does not carry reads as zero. */ + v = (i == IDX_A) ? 0xff : 0; + } else { + v = (pixel >> f->shift[i]) & ((1u << bits) - 1u); + v = bit_expand(v, bits); + } + argb |= v << (24 - 8 * i); + } + return argb; +} + +void xpsb_pix_store(const struct xpsb_format *f, uint8_t *p, uint32_t argb) +{ + uint32_t pixel = 0; + unsigned int i; + + for (i = 0; i < 4; i++) { + unsigned int bits = f->bits[i]; + uint32_t v = (argb >> (24 - 8 * i)) & 0xff; + + if (bits) + pixel |= (v >> (8 - bits)) << f->shift[i]; + } + raw_store(f, p, pixel); +} + +/* Xpsb_clampGL is clamped the same way: neither psbExaSrfInfo nor psb_video.c + * ever selects it, so its border behaviour is unobserved. */ +static int wrap_coord(int x, int n, int mode) +{ + if (mode == XPSB_ADDR_REPEAT) { + x %= n; + if (x < 0) + x += n; + return x; + } + if (x < 0) + return 0; + if (x >= n) + return n - 1; + return x; +} + +static uint32_t texel(const struct xpsb_image *img, int x, int y) +{ + x = wrap_coord(x, (int)img->w, img->umode); + y = wrap_coord(y, (int)img->h, img->vmode); + return xpsb_pix_load(&img->fmt, + img->base + (ptrdiff_t)y * img->stride + + (ptrdiff_t)x * (img->fmt.bpp >> 3)); +} + +static uint32_t lerp_argb(uint32_t a, uint32_t b, int w) +{ + uint32_t out = 0; + int i; + + for (i = 0; i < 4; i++) { + int shift = 8 * i; + int ca = (a >> shift) & 0xff, cb = (b >> shift) & 0xff; + + out |= (uint32_t)(ca + (((cb - ca) * w + 128) >> 8)) << shift; + } + return out; +} + +uint32_t xpsb_image_sample(const struct xpsb_image *img, float u, float v) +{ + float fx, fy; + int x, y, wx, wy; + + if (img->filter == XPSB_FILTER_NEAREST) + return texel(img, (int)floorf(u * (float)img->w), + (int)floorf(v * (float)img->h)); + + fx = u * (float)img->w - 0.5f; + fy = v * (float)img->h - 0.5f; + x = (int)floorf(fx); + y = (int)floorf(fy); + wx = (int)((fx - (float)x) * 256.0f + 0.5f); + wy = (int)((fy - (float)y) * 256.0f + 0.5f); + + return lerp_argb(lerp_argb(texel(img, x, y), texel(img, x + 1, y), wx), + lerp_argb(texel(img, x, y + 1), + texel(img, x + 1, y + 1), wx), wy); +} diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/xpsb_pixel.h mesa-26.2.2/src/gallium/drivers/sgx/xpsb_pixel.h --- mesa-26.2.2.orig/src/gallium/drivers/sgx/xpsb_pixel.h 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/xpsb_pixel.h 2026-09-08 10:57:36.684811533 +0200 @@ -0,0 +1,50 @@ +/* Open replacement for Xpsb.so - PICT format handling and texture sampling. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#ifndef _XPSB_PIXEL_H_ +#define _XPSB_PIXEL_H_ + +#include + +/* Addressing modes, same numbering as XpsbAddrModes in Xpsb.h. */ +#define XPSB_ADDR_REPEAT 0 +#define XPSB_ADDR_CLAMP 1 +#define XPSB_ADDR_CLAMPGL 2 + +#define XPSB_FILTER_NEAREST 0 +#define XPSB_FILTER_LINEAR 1 + +/* A PICT format broken into shifts and widths. Only the A, ARGB and ABGR + * types occur in the paths the DDX drives; anything else is rejected so the + * caller can decline the operation. */ +struct xpsb_format { + unsigned int bpp; + unsigned int bits[4]; /* a, r, g, b */ + unsigned int shift[4]; + unsigned int has_alpha; +}; + +/* A mapped surface: pixels, geometry, format and sampler state. */ +struct xpsb_image { + uint8_t *base; + int stride; + unsigned int w, h; + struct xpsb_format fmt; + int filter; + int umode, vmode; +}; + +int xpsb_format_parse(unsigned int pict_format, struct xpsb_format *f); + +uint32_t xpsb_pix_load(const struct xpsb_format *f, const uint8_t *p); +void xpsb_pix_store(const struct xpsb_format *f, uint8_t *p, uint32_t argb); + +/* Sample at normalised coordinates, honouring filter and addressing mode. + * Returns premultiplied ARGB8888, alpha forced to 0xff for formats without + * an alpha channel, exactly as psbPixelARGB8888 in the DDX does. */ +uint32_t xpsb_image_sample(const struct xpsb_image *img, float u, float v); + +#endif /* _XPSB_PIXEL_H_ */ diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/xpsb_shader.c mesa-26.2.2/src/gallium/drivers/sgx/xpsb_shader.c --- mesa-26.2.2.orig/src/gallium/drivers/sgx/xpsb_shader.c 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/xpsb_shader.c 2026-09-08 10:57:36.684818893 +0200 @@ -0,0 +1,102 @@ +/* Open replacement for Xpsb.so - composite pixel shader generation. + * + * Transcribed from XpsbCompositeShader at 0xa660 and the two tables it + * indexes, at .rodata 0xd700 (per-op precompiled program, all zero in this + * build) and 0xd740 (per-op offset into the blend-word blob at 0xd0e0). + * + * A Render composite is two USSE SOP2 instructions: + * + * sop2 i0, , , s2a, zero, add + asop2 s2a, zero, add + * sop2.end o0, i0, o0, , , add + asop2 , , add + * + * The first modulates the source by the mask alpha into internal register + * i0; the second applies the Porter-Duff factor pair for the operator. Both + * were checked against tools/isa-usse, which reproduces every word. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#include "xpsb_shader.h" + +/* The per-operator blend instruction, from the blob at .rodata 0xd0e0 offset + * 0x30 upwards. Disassembled, these are exactly the Porter-Duff factor pairs + * the Render specification calls for: + * + * Clear mov o0, #0 + * Src src*1 + dst*0 + * Dst src*0 + dst*1 + * Over src*1 + dst*(1-Sa) + * OverReverse src*(1-Da) + dst*1 + * In src*Da + dst*0 + * InReverse src*0 + dst*Sa + * Out src*(1-Da) + dst*0 + * OutReverse src*0 + dst*(1-Sa) + * Atop src*Da + dst*(1-Sa) + * AtopReverse src*(1-Da) + dst*Sa + * Xor src*(1-Da) + dst*(1-Sa) + * Add src*1 + dst*1 + * Saturate src*asat + dst*1 + * + * s1a is the source alpha and s2a the destination alpha, because the blend + * instruction's sources are (i0, o0) in that order. + */ +static const struct { uint32_t lo, hi; } op_blend[XPSB_NUM_OPS] = { + { 0x00000000, 0xfca40001 }, /* Clear */ + { 0xd0000000, 0x81860801 }, /* Src */ + { 0xd0000000, 0x80868005 }, /* Dst */ + { 0xd0000000, 0x81868a1d }, /* Over */ + { 0xd0000000, 0x81a68905 }, /* OverReverse */ + { 0xd0000000, 0x80a60101 }, /* In */ + { 0xd0000000, 0x80860219 }, /* InReverse */ + { 0xd0000000, 0x81a60901 }, /* Out */ + { 0xd0000000, 0x8086821d }, /* OutReverse */ + { 0xd0000000, 0x80a6831d }, /* Atop */ + { 0xd0000000, 0x81a60b19 }, /* AtopReverse */ + { 0xd0000000, 0x81a68b1d }, /* Xor */ + { 0xd0000000, 0x81868805 }, /* Add */ + { 0xd0000000, 0x80868945 }, /* Saturate */ +}; + +/* PictOpClear and PictOpSrc do not read the destination. */ +const uint8_t xpsb_opaque_ops[XPSB_NUM_OPS] = { 1, 1 }; + +int xpsb_composite_shader(int op, const struct xpsb_shader_flags *f, + uint32_t *out) +{ + uint32_t w0, w1; + + if (op < 0 || op >= XPSB_NUM_OPS) + return -1; + + if (f->src_is_a8) { + /* An a8 source contributes alpha only: i0.rgb is forced to + * zero and only the alpha half of the pair does any work. */ + w0 = f->scalar_mask ? 0xb0000000u : 0xa0000001u; + w1 = 0x80e80003u; + } else { + w0 = 0xa0000001u; + w1 = f->src_no_alpha ? 0x80c80107u : 0x80e80103u; + if (f->scalar_mask) + w0 = (w0 & 0x0ffffff0u) | 0xb0000000u; + else if (f->scalar_src) + w0 = (w0 & 0x0ffffff0u) | 0xe0000000u; + } + + /* PictOpSrc needs no destination read, so the modulate writes the + * output register and ends the program; the blend below is then dead + * but is still emitted, and computes the same value. */ + if (op == XPSB_OP_SRC) + w1 = ((w1 & 0xffbbffffu) | 0x40000u) & 0xfff7fffdu; + /* PictOpSaturate is the one operator whose factor is a clamp rather + * than a product, so the modulate becomes the write-masked form with + * a min combiner. */ + else if (op == XPSB_OP_SATURATE) + w1 = (w1 & 0xeffffeffu) | 0x10000000u; + + out[0] = w0; + out[1] = w1; + out[2] = op_blend[op].lo; + out[3] = op_blend[op].hi; + return 0; +} diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/xpsb_shader.h mesa-26.2.2/src/gallium/drivers/sgx/xpsb_shader.h --- mesa-26.2.2.orig/src/gallium/drivers/sgx/xpsb_shader.h 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/xpsb_shader.h 2026-09-08 10:57:36.685014832 +0200 @@ -0,0 +1,46 @@ +/* Open replacement for Xpsb.so - composite pixel shader generation. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#ifndef _XPSB_SHADER_H_ +#define _XPSB_SHADER_H_ + +#include + +#define XPSB_NUM_OPS 14 + +#define XPSB_OP_CLEAR 0 +#define XPSB_OP_SRC 1 +#define XPSB_OP_DST 2 +#define XPSB_OP_OVER 3 +#define XPSB_OP_ADD 12 +#define XPSB_OP_SATURATE 13 + +/* How the source and mask reach the shader. The binary derives all four from + * the picture formats in psb3DPrepareComposite: + * + * src_no_alpha the source format has no alpha channel, so its alpha is + * taken as one rather than sampled + * src_is_a8 the source is PICT_a8 (0x08018000), alpha only + * scalar_src the source is a repeating 1x1 picture, supplied as a + * constant in a secondary attribute instead of a texture + * scalar_mask likewise for the mask; also set when there is no mask + */ +struct xpsb_shader_flags { + int src_no_alpha; + int src_is_a8; + int scalar_src; + int scalar_mask; +}; + +/* Emit the two SOP2 instructions for one Render operator into out[4]. + * Returns -1 for an operator outside 0..13. */ +int xpsb_composite_shader(int op, const struct xpsb_shader_flags *f, + uint32_t *out); + +/* Non-zero for operators that never read the destination. */ +extern const uint8_t xpsb_opaque_ops[XPSB_NUM_OPS]; + +#endif /* _XPSB_SHADER_H_ */ diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/xpsb_vidshader.c mesa-26.2.2/src/gallium/drivers/sgx/xpsb_vidshader.c --- mesa-26.2.2.orig/src/gallium/drivers/sgx/xpsb_vidshader.c 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/xpsb_vidshader.c 2026-09-08 10:57:36.685023952 +0200 @@ -0,0 +1,247 @@ +/* Open replacement for Xpsb.so - the GPU YUV to RGB video shaders. + * + * The closed module does not generate USSE code for the video path: XpsbInit + * uploads one fixed 440-byte image from .rodata into a buffer object and + * psbBlitYUV selects one of three programs inside it by FOURCC. The image is + * reproduced here verbatim, together with the two unrelated routes by which + * the conversion coefficients reach the hardware. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#include + +#include "xpsb_vidshader.h" +#include "xpsb_yuv.h" + +/* Extracted from upstream/blobs/Xpsb.so, ELF32 i386: .rodata has VMA 0xcd80 at + * file offset 0xcd80 (objdump -h), so the image XpsbInit passes to XpsbBOData + * as (.rodata + 0xd0e0, 0x1b8) is file bytes 0xd0e0 .. 0xd297. + * + * The two words of each instruction are stored low word first; the tool in + * tools/isa-usse takes them as hi<<32 | lo. + * + * 0x000 .. 0x10f 17 x 16-byte composite programs, indexed by + * XpsbCompositeShader through the .rodata table at 0xd740. + * Not part of the video path; carried because the whole image + * is uploaded as one buffer. + * + * 0x110 packed YUV (UYVY / YUY2), 14 instructions, disassembled clean by + * tools/isa-usse/usse-dis -x: + * + * 0: 0x400015BCA0400002 unpckf32o8 r2, pa0.0, pa2.0, nearest + * 1: 0x400015BCA0214002 unpckf32o8 r1, pa0.1, pa2.1, nearest + * 2: 0x400011BCA0028002 unpckf32u8 r0, pa0.2, pa2.2, nearest + * 3: 0x0002100070001A09 fmad r0, r0, c52, sa9 + * 4: 0x380B0E00F0600003 efo.nosched r3= i0, i0 = m0, i1 = m1, + * a0=src0+src1 a1=src2+src0, + * m0=src0*src1, m1=src0*src2, r0, sa0, sa3 + * 5: 0x38220E00F0604084 efo.nosched r3= a0, i0 = a0, i1 = a1, + * a0=i0+m0, a1=i1+m1, + * m0=src0*src1, m1=src0*src2, r1, sa1, sa4 + * 6: 0x38220E00F0608105 efo.nosched r3= a0, i0 = a0, i1 = a1, + * a0=i0+m0, a1=i1+m1, + * m0=src0*src1, m1=src0*src2, r2, sa2, sa5 + * 7: 0x381B0E00F0800306 efo.nosched r4= i1, i0 = m0, i1 = m1, + * a0=src0+src1 a1=src2+src0, + * m0=src0*src1, m1=src0*src2, r0, sa6, sa6 + * 8: 0x00011000F0A04380 fmad r5, r1, sa7, i0 + * 9: 0x00001000C0A08405 fmad r5, r2, sa8, r5 + * 10: 0x40801C1100040183 pcku8f32.skipinv.scale o0.bytemask0100, r3.0, r3.0 + * 11: 0x40801C0900040204 pcku8f32.skipinv.scale o0.bytemask0010, r4.0, r4.0 + * 12: 0x40841C0500040285 pcku8f32.skipinv.end.scale o0.bytemask0001, r5.0, r5.0 + * 13: 0xF800014000000000 nop + * + * The texture unit delivers one packed texel in pa0/pa2 with byte 2 = + * Y, byte 1 = Cb, byte 0 = Cr. 0..2 unpack them, signed (b - 128) for + * chroma and unsigned for luma; 3 biases luma by sa9; 4..9 are a row + * major 3x3 matrix-vector product with the matrix in sa0..sa8; 10..12 + * pack R,G,B into bytes 2,1,0 of o0 and leave alpha alone. That is + * conversionData[0..9] exactly. sa10 (videoGamma) is loaded and unused. + * c52 is a hardware constant register whose value is NOT DETERMINED + * from the binary; the DDX arithmetic only works out if it is 1.0f. + * + * 0x180 NV12 and 0x1a0 YV12 / I420, three instructions each. usse-dis + * classifies all six words as opcode 0x17 = FIRH and has no printer for + * that opcode, so it prints them as raw words; its trailing FIRH + * annotation is a C comment, rewritten as (FIRH) to nest here: + * + * 0: 0xB8A04841A0000081 .word 0xB8A04841, 0xA0000081 (FIRH) NV12 + * 1: 0xB8904849A0000081 .word 0xB8904849, 0xA0000081 (FIRH) + * 2: 0xB8844851A0000081 .word 0xB8844851, 0xA0000081 (FIRH) + * 3: 0xF800014000000000 nop + * + * 0: 0xB8A04841A0000101 .word 0xB8A04841, 0xA0000101 (FIRH) YV12/I420 + * 1: 0xB8904849A0000101 .word 0xB8904849, 0xA0000101 (FIRH) + * 2: 0xB8844851A0000101 .word 0xB8844851, 0xA0000101 (FIRH) + * + * No decode is invented here. What the standard field layout gives is + * that all three write o0, the third carries .end (which is why the + * YV12 program needs no trailing nop), and w1[6:3] takes the values + * 8, 9, 10 - one per output channel, matching the three coefficient + * groups psb_pack_coeffs builds. The FIRH tap layout and the binding to + * the coefficient registers of the planar route below are NOT + * DETERMINED; USSE_ISA.txt documents only the opcode and repeat fields. + */ +const uint32_t xpsb_usse_blob[XPSB_USSE_BLOB_DWORDS] = { + /* 0x000: composite programs */ + 0xa0000000, 0x28851001, 0x00000000, 0xf8000140, + 0xa01f8000, 0x50851439, 0x00000000, 0xf8000140, + 0xa01f8000, 0x50851431, 0x00000000, 0xf8000140, + 0x00000000, 0xfca40001, 0x00000000, 0xf8000140, + 0xd0000000, 0x81860801, 0x00000000, 0xf8000140, + 0xd0000000, 0x80868005, 0x00000000, 0xf8000140, + 0xd0000000, 0x81868a1d, 0x00000000, 0xf8000140, + 0xd0000000, 0x81a68905, 0x00000000, 0xf8000140, + 0xd0000000, 0x80a60101, 0x00000000, 0xf8000140, + 0xd0000000, 0x80860219, 0x00000000, 0xf8000140, + 0xd0000000, 0x81a60901, 0x00000000, 0xf8000140, + 0xd0000000, 0x8086821d, 0x00000000, 0xf8000140, + 0xd0000000, 0x80a6831d, 0x00000000, 0xf8000140, + 0xd0000000, 0x81a60b19, 0x00000000, 0xf8000140, + 0xd0000000, 0x81a68b1d, 0x00000000, 0xf8000140, + 0xd0000000, 0x81868805, 0x00000000, 0xf8000140, + 0xd0000000, 0x80868945, 0x00000000, 0xf8000140, + /* 0x110: packed YUV, UYVY / YUY2 */ + 0xa0400002, 0x400015bc, 0xa0214002, 0x400015bc, + 0xa0028002, 0x400011bc, 0x70001a09, 0x00021000, + 0xf0600003, 0x380b0e00, 0xf0604084, 0x38220e00, + 0xf0608105, 0x38220e00, 0xf0800306, 0x381b0e00, + 0xf0a04380, 0x00011000, 0xc0a08405, 0x00001000, + 0x00040183, 0x40801c11, 0x00040204, 0x40801c09, + 0x00040285, 0x40841c05, 0x00000000, 0xf8000140, + /* 0x180: NV12 */ + 0xa0000081, 0xb8a04841, 0xa0000081, 0xb8904849, + 0xa0000081, 0xb8844851, 0x00000000, 0xf8000140, + /* 0x1a0: YV12 / I420 */ + 0xa0000101, 0xb8a04841, 0xa0000101, 0xb8904849, + 0xa0000101, 0xb8844851, +}; + +uint32_t xpsb_vidshader_offset(uint32_t fourcc) +{ + switch (fourcc) { + case XPSB_FOURCC_UYVY: + case XPSB_FOURCC_YUY2: + return XPSB_USSE_OFF_PACKED; + case XPSB_FOURCC_NV12: + return XPSB_USSE_OFF_NV12; + case XPSB_FOURCC_YV12: + case XPSB_FOURCC_I420: + return XPSB_USSE_OFF_PLANAR; + default: + return ~0u; + } +} + +const uint32_t *xpsb_vidshader(uint32_t fourcc, unsigned *n_dwords) +{ + uint32_t off = xpsb_vidshader_offset(fourcc); + unsigned len; + + if (off == ~0u) + return NULL; + + if (off == XPSB_USSE_OFF_PACKED) + len = XPSB_USSE_LEN_PACKED; + else if (off == XPSB_USSE_OFF_NV12) + len = XPSB_USSE_LEN_NV12; + else + len = XPSB_USSE_LEN_PLANAR; + + if (n_dwords) + *n_dwords = len / 4; + return &xpsb_usse_blob[off / 4]; +} + +/* Attribute DMA descriptor: bits 7:0 dwords - 1, bits 24:21 burst - 1, bit 31 + * last chunk. The n > 16 chunking path of the closed module is never reached + * by the video path (n is 11 or 0) and is not reproduced. */ +uint32_t xpsb_vidshader_sa_dma_ctrl(unsigned n) +{ + if (n < 2 || n > 16) + return 0; + return 0x80000000u | ((n - 1) << 21) | (n - 1); +} + +unsigned xpsb_vidshader_emit_sa_consts(uint32_t *out, + const float *conversion_data, + unsigned n) +{ + if (!out || !conversion_data) + return 0; + memcpy(out, conversion_data, n * sizeof(float)); + return n; +} + +/* const_addr is the GPU address of the constant block emitted above; in a live + * command buffer that dword is a relocation the kernel patches. */ +int xpsb_vidshader_emit_sa_pds(struct xpsb_vidshader_sa_pds *pds, + const float *conversion_data, unsigned n, + uint32_t const_addr) +{ + if (!pds || n > 16) + return -1; + + memset(pds, 0, sizeof(*pds)); + + if (n == 0) { + pds->prog[8] = XPSB_PDS_HALT; + pds->n_prog = 9; + pds->data_dwords = 8; + pds->n_attrs = 1; + return 0; + } + + if (n == 1) { + if (!conversion_data) + return -1; + memcpy(&pds->prog[0], conversion_data, sizeof(float)); + pds->prog[8] = XPSB_PDS_DOUTA; + pds->prog[9] = XPSB_PDS_HALT; + pds->n_prog = 10; + pds->data_dwords = 8; + pds->n_attrs = 1; + return 0; + } + + pds->prog[0] = const_addr; + pds->prog[1] = xpsb_vidshader_sa_dma_ctrl(n); + pds->prog[12] = XPSB_PDS_DOUTD; + pds->prog[13] = XPSB_PDS_HALT; + pds->n_prog = 14; + pds->data_dwords = 12; + pds->n_attrs = n; + return 0; +} + +/* The nine sgx_coeffs words go to hardware filter coefficient registers, one + * group of three per output channel; the B group is not contiguous with R and + * G. Symbolic names for these registers are NOT DETERMINED - they are absent + * from the GPL psb_reg.h and would need Imagination's SGX535 sgxdefs.h. The + * leading 0x0a9c = 3 is written unconditionally; group count, tap count and + * filter enable are all consistent with it and nothing distinguishes them. */ +const uint32_t xpsb_vidshader_coeff_regs[XPSB_VIDSHADER_COEFF_PAIRS] = { + 0x0a9c, + 0x0a84, 0x0a88, 0x0a8c, /* R */ + 0x0a90, 0x0a94, 0x0a98, /* G */ + 0x0b20, 0x0b24, 0x0b28, /* B */ +}; + +unsigned xpsb_vidshader_emit_coeff_regs(uint32_t *out, + const uint32_t *sgx_coeffs) +{ + unsigned i; + + if (!out || !sgx_coeffs) + return 0; + + out[0] = xpsb_vidshader_coeff_regs[0]; + out[1] = 3; + for (i = 0; i < 9; i++) { + out[2 + 2 * i] = xpsb_vidshader_coeff_regs[1 + i]; + out[3 + 2 * i] = sgx_coeffs[i]; + } + return XPSB_VIDSHADER_COEFF_DWORDS; +} diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/xpsb_vidshader.h mesa-26.2.2/src/gallium/drivers/sgx/xpsb_vidshader.h --- mesa-26.2.2.orig/src/gallium/drivers/sgx/xpsb_vidshader.h 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/xpsb_vidshader.h 2026-09-08 10:57:36.685031213 +0200 @@ -0,0 +1,70 @@ +/* Open replacement for Xpsb.so - the GPU YUV to RGB video shaders. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#ifndef _XPSB_VIDSHADER_H_ +#define _XPSB_VIDSHADER_H_ + +#include + +/* The "USSE static buffer" the closed module uploads once from .rodata 0xd0e0, + * 440 bytes, of which the video path uses the three programs below. */ +#define XPSB_USSE_BLOB_BYTES 0x1b8u +#define XPSB_USSE_BLOB_DWORDS (XPSB_USSE_BLOB_BYTES / 4) + +#define XPSB_USSE_OFF_COMPOSITE 0x000u /* 17 x 16-byte composite programs */ +#define XPSB_USSE_OFF_PACKED 0x110u /* UYVY / YUY2, 14 instructions */ +#define XPSB_USSE_OFF_NV12 0x180u /* NV12, 3 instructions + nop pad */ +#define XPSB_USSE_OFF_PLANAR 0x1a0u /* YV12 / I420, 3 instructions */ + +#define XPSB_USSE_LEN_PACKED 0x70u +#define XPSB_USSE_LEN_NV12 0x20u +#define XPSB_USSE_LEN_PLANAR 0x18u + +extern const uint32_t xpsb_usse_blob[XPSB_USSE_BLOB_DWORDS]; + +/* Returns the program for one of the five video FOURCCs, NULL otherwise; the + * blob has no default case, so an unknown FOURCC has to be rejected here. */ +const uint32_t *xpsb_vidshader(uint32_t fourcc, unsigned *n_dwords); + +/* Byte offset of that program within the uploaded blob, or ~0u. */ +uint32_t xpsb_vidshader_offset(uint32_t fourcc); + +/* Packed route: conversionData[11] verbatim into the command buffer, then a + * PDS DMA into secondary attributes sa0..sa10. */ +#define XPSB_VIDSHADER_SA_PACKED 11u +#define XPSB_VIDSHADER_SA_PDS_MAX 14u +#define XPSB_PDS_DOUTD 0x07030223u +#define XPSB_PDS_DOUTA 0x07030226u +#define XPSB_PDS_HALT 0xaf000000u + +struct xpsb_vidshader_sa_pds { + uint32_t prog[XPSB_VIDSHADER_SA_PDS_MAX]; + unsigned n_prog; /* dwords of prog[] to emit */ + unsigned data_dwords; /* pixel shader descriptor slot [5] */ + unsigned n_attrs; /* pixel shader descriptor slot [6] */ +}; + +uint32_t xpsb_vidshader_sa_dma_ctrl(unsigned n); + +unsigned xpsb_vidshader_emit_sa_consts(uint32_t *out, + const float *conversion_data, + unsigned n); + +int xpsb_vidshader_emit_sa_pds(struct xpsb_vidshader_sa_pds *pds, + const float *conversion_data, unsigned n, + uint32_t const_addr); + +/* Planar route: sgx_coeffs[9] as ten (register, value) pairs appended to the + * 3D raster register list. */ +#define XPSB_VIDSHADER_COEFF_PAIRS 10u +#define XPSB_VIDSHADER_COEFF_DWORDS (2u * XPSB_VIDSHADER_COEFF_PAIRS) + +extern const uint32_t xpsb_vidshader_coeff_regs[XPSB_VIDSHADER_COEFF_PAIRS]; + +unsigned xpsb_vidshader_emit_coeff_regs(uint32_t *out, + const uint32_t *sgx_coeffs); + +#endif /* _XPSB_VIDSHADER_H_ */ diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/xpsb_xhw.c mesa-26.2.2/src/gallium/drivers/sgx/xpsb_xhw.c --- mesa-26.2.2.orig/src/gallium/drivers/sgx/xpsb_xhw.c 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/xpsb_xhw.c 2026-09-08 10:57:36.685040043 +0200 @@ -0,0 +1,161 @@ +/* Open replacement for Xpsb.so - XHW responder loop. + * + * Transcribed from XpsbThread at 0x30f0. The kernel delegates scene + * management to userspace and refuses to submit 3D work without a responder + * ("No Xpsb 3D extension available"), so this loop is what makes the GPL + * driver usable at all. + * + * The protocol is one shared page: DRM_PSB_XHW blocks until the kernel has + * copied a request into it, and the same ioctl, called again, first delivers + * whatever reply the page now holds. So a reply is posted simply by writing + * it back in place and going round the loop. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#include +#include +#include +#include + +#include + +#include "xpsb_xhw.h" + +int xpsb_xhw_dispatch(struct xpsb_xhw *ctl) +{ + volatile struct drm_psb_xhw_arg *xa = ctl->shared; + struct xpsb_hw *hw = &ctl->hw; + uint32_t cookie[PSB_HW_COOKIE_SIZE]; + uint32_t oom_cmds[PSB_HW_OOM_CMD_SIZE]; + uint32_t size, clear_start, clear_pages, value, bca, rca, oflags; + int i; + + switch (xa->op) { + case PSB_XHW_FIRE_RASTER: + if (xa->arg.sb.fire_flags & 0x10000000) { + xa->ret = xpsb_kick_ta(hw); + } else { + xpsb_wr(hw, SGX_DPM_STATE, 0); + xpsb_wr(hw, SGX_DPM_CONTEXT, 0); + xa->ret = xpsb_kick_render(hw); + } + break; + + case PSB_XHW_SCENE_INFO: + memcpy(cookie, (const void *)xa->cookie, sizeof cookie); + xa->ret = xpsb_scene_info(hw, xa->arg.si.w, xa->arg.si.h, + cookie, &size, &clear_start, + &clear_pages); + memcpy((void *)xa->cookie, cookie, sizeof cookie); + xa->arg.si.size = size; + xa->arg.si.clear_p_start = clear_start; + xa->arg.si.clear_num_pages = clear_pages; + break; + + case PSB_XHW_SCENE_BIND_FIRE: + memcpy(cookie, (const void *)xa->cookie, sizeof cookie); + memcpy(oom_cmds, (const void *)xa->arg.sb.oom_cmds, + sizeof oom_cmds); + rca = xa->arg.sb.rca; + xa->ret = xpsb_scene_switch_fire(hw, xa->arg.sb.fire_flags, + xa->arg.sb.hw_context, cookie, + xa->arg.sb.offset, + xa->arg.sb.engine, + xa->arg.sb.flags, oom_cmds, + xa->arg.sb.num_oom_cmds, &rca); + memcpy((void *)xa->cookie, cookie, sizeof cookie); + xa->arg.sb.rca = rca; + break; + + case PSB_XHW_TA_MEM_INFO: + memcpy(cookie, (const void *)xa->cookie, sizeof cookie); + xa->ret = xpsb_ta_mem_info(xa->arg.bi.pages, cookie, &size); + memcpy((void *)xa->cookie, cookie, sizeof cookie); + xa->arg.bi.size = size; + break; + + case PSB_XHW_RESET_DPM: + xpsb_sgx_uninit(hw); + xpsb_sgx_initialize(hw); + break; + + case PSB_XHW_OOM: + memcpy(cookie, (const void *)xa->cookie, sizeof cookie); + xpsb_oom_abort(hw, cookie, &bca, &rca, &oflags); + memcpy((void *)xa->cookie, cookie, sizeof cookie); + xa->arg.oom.bca = bca; + xa->arg.oom.rca = rca; + xa->arg.oom.flags = oflags; + break; + + case PSB_XHW_TERMINATE: + /* Flush any reply still sitting in the page before the loop + * stops calling the ioctl that would have delivered it. */ + if (xa->issue_irq) + xpsb_wrb(hw, SGX_EVENT_STATUS, SGX_EV_SW_EVENT); + return 1; + + case PSB_XHW_VISTEST: + for (i = 0; i < PSB_HW_FEEDBACK_SIZE; i++) + xa->arg.feedback[i] = + xpsb_rd(hw, SGX_VISTEST_RESULT + 4 * i); + break; + + case PSB_XHW_RESUME: + xpsb_sgx_initialize(hw); + xa->ret = 0; + break; + + case PSB_XHW_TA_MEM_LOAD: + memcpy(cookie, (const void *)xa->cookie, sizeof cookie); + xa->ret = xpsb_ta_mem_load(hw, xa->arg.bl.pt_offset, + xa->arg.bl.param_offset, + xa->arg.bl.flags, cookie); + memcpy((void *)xa->cookie, cookie, sizeof cookie); + break; + + case PSB_XHW_CHECK_LOCKUP: + case PSB_XHW_HOTPLUG: + xa->ret = xpsb_check_lockup(hw, &value); + xa->arg.cl.value = value; + break; + + default: + break; + } + return 0; +} + +void *xpsb_xhw_thread(void *arg) +{ + struct xpsb_xhw *ctl = arg; + sigset_t set; + int ret; + + /* Leave the signals a debugger or a fault needs deliverable and block + * everything else, so X's own handlers are never run on this thread. */ + sigfillset(&set); + sigdelset(&set, SIGFPE); + sigdelset(&set, SIGILL); + sigdelset(&set, SIGSEGV); + sigdelset(&set, SIGBUS); + pthread_sigmask(SIG_BLOCK, &set, NULL); + + for (;;) { + ret = drmCommandNone(ctl->fd, DRM_PSB_XHW); + if (ret == -EAGAIN) + continue; + if (ret) + continue; + if (ctl->shared->op > PSB_XHW_HOTPLUG) + continue; + if (xpsb_xhw_dispatch(ctl)) + break; + } + + if (ctl->exit_flag) + *ctl->exit_flag = 1; + return NULL; +} diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/xpsb_xhw.h mesa-26.2.2/src/gallium/drivers/sgx/xpsb_xhw.h --- mesa-26.2.2.orig/src/gallium/drivers/sgx/xpsb_xhw.h 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/xpsb_xhw.h 2026-09-08 10:57:36.685047693 +0200 @@ -0,0 +1,94 @@ +/* Open replacement for Xpsb.so - XHW wire ABI and responder. + * + * The structures below mirror psb_drm.h from the GPL kernel module; they are + * the ioctl ABI, not implementation, and are reproduced so this module can be + * built without the kernel source tree in the include path. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#ifndef _XPSB_XHW_H_ +#define _XPSB_XHW_H_ + +#include + +#include "xpsb_hw.h" + +#define DRM_PSB_XHW_INIT 0x01 +#define DRM_PSB_XHW 0x02 + +#define PSB_XHW_INIT 0x00 +#define PSB_XHW_TAKEDOWN 0x01 + +#define PSB_XHW_FIRE_RASTER 0x00 +#define PSB_XHW_SCENE_INFO 0x01 +#define PSB_XHW_SCENE_BIND_FIRE 0x02 +#define PSB_XHW_TA_MEM_INFO 0x03 +#define PSB_XHW_RESET_DPM 0x04 +#define PSB_XHW_OOM 0x05 +#define PSB_XHW_TERMINATE 0x06 +#define PSB_XHW_VISTEST 0x07 +#define PSB_XHW_RESUME 0x08 +#define PSB_XHW_TA_MEM_LOAD 0x09 +#define PSB_XHW_CHECK_LOCKUP 0x0a +#define PSB_XHW_HOTPLUG 0x0b + +#define PSB_HW_COOKIE_SIZE 16 +#define PSB_HW_FEEDBACK_SIZE 8 +#define PSB_HW_OOM_CMD_SIZE 6 + +struct drm_psb_xhw_init_arg { + uint32_t operation; + uint32_t buffer_handle; + uint32_t tmpBOHandle; + void *fbPhys; + uint32_t fbSize; +}; + +struct drm_psb_xhw_arg { + uint32_t op; + int ret; + uint32_t irq_op; + uint32_t issue_irq; + uint32_t cookie[PSB_HW_COOKIE_SIZE]; + union { + struct { + uint32_t w, h, size, clear_p_start, clear_num_pages; + } si; + struct { + uint32_t fire_flags, hw_context, offset, engine, flags; + uint32_t rca, num_oom_cmds; + uint32_t oom_cmds[PSB_HW_OOM_CMD_SIZE]; + } sb; + struct { + uint32_t pages, size; + } bi; + struct { + uint32_t bca, rca, flags; + } oom; + struct { + uint32_t pt_offset, param_offset, flags; + } bl; + struct { + uint32_t value; + } cl; + uint32_t feedback[PSB_HW_FEEDBACK_SIZE]; + } arg; +}; + +struct xpsb_xhw { + int fd; + struct xpsb_hw hw; + volatile struct drm_psb_xhw_arg *shared; + volatile int *exit_flag; +}; + +/* Answer one request already present in ctl->shared. Returns non-zero when + * the responder has been asked to terminate. */ +int xpsb_xhw_dispatch(struct xpsb_xhw *ctl); + +/* Blocking responder loop; runs until TERMINATE or *exit_flag. */ +void *xpsb_xhw_thread(void *arg); + +#endif /* _XPSB_XHW_H_ */ diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/xpsb_yuv.c mesa-26.2.2/src/gallium/drivers/sgx/xpsb_yuv.c --- mesa-26.2.2.orig/src/gallium/drivers/sgx/xpsb_yuv.c 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/xpsb_yuv.c 2026-09-08 10:57:36.685056335 +0200 @@ -0,0 +1,222 @@ +/* Open replacement for Xpsb.so - YUV to RGB video blit. + * + * psbDisplayVideo in the DDX ignores the return value of psbBlitYUV, so this + * path has to produce pixels rather than decline: an unimplemented blit shows + * as a video window that stays blank. Sampling and the colour space transform + * follow the same coefficients the DDX prepares for the 3D pipeline, so the + * output matches what the closed module would have drawn. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#include +#include + +#include "xpsb_yuv.h" + +/* The chroma bias is not part of conversion_data - the DDX folds only the + * luma offset into element 9 and leaves the fixed -128 to the shader. */ +#define CHROMA_BIAS 128.0f + +void xpsb_yuv_conv_packed(const float *conversion_data, + struct xpsb_yuv_conv *conv) +{ + float y_off = conversion_data[9]; + int i, j; + + for (i = 0; i < 3; i++) { + const float *m = &conversion_data[3 * i]; + float bias = m[0] * y_off - + CHROMA_BIAS * (m[1] + m[2]); + + for (j = 0; j < 3; j++) + conv->c[i][j] = (int32_t)lrintf(255.0f * 256.0f * m[j]); + conv->k[i] = (int32_t)lrintf(255.0f * 256.0f * bias); + conv->shift[i] = 8; + } +} + +void xpsb_yuv_conv_planar(const uint32_t *sgx_coeffs, struct xpsb_yuv_conv *conv) +{ + int i; + + for (i = 0; i < 3; i++) { + const uint32_t *w = &sgx_coeffs[3 * i]; + + conv->c[i][0] = (int8_t)((w[0] >> 24) & 0xff); + conv->c[i][1] = (int8_t)(w[0] & 0xff); + conv->c[i][2] = (int8_t)((w[1] >> 8) & 0xff); + conv->k[i] = (int16_t)((w[2] >> 4) & 0xffff); + conv->shift[i] = (int)(w[2] & 0xf); + } +} + +/* BT.601 limited range, used when the DDX passes no coefficients at all. */ +void xpsb_yuv_conv_default(struct xpsb_yuv_conv *conv) +{ + static const float bt601[11] = { + 1.0f, 0.0f, 1.4075f, + 1.0f, -0.3455f, -0.7169f, + 1.0f, 1.7790f, 0.0f, + -16.0f, 0.0f + }; + float scaled[11]; + int i; + + for (i = 0; i < 9; i++) + scaled[i] = bt601[i] / 219.0f; + scaled[9] = bt601[9]; + scaled[10] = bt601[10]; + xpsb_yuv_conv_packed(scaled, conv); +} + +int xpsb_yuv_planes(uint32_t fourcc) +{ + switch (fourcc) { + case XPSB_FOURCC_YUY2: + case XPSB_FOURCC_UYVY: + return 1; + case XPSB_FOURCC_NV12: + return 2; + case XPSB_FOURCC_YV12: + case XPSB_FOURCC_I420: + return 3; + default: + return 0; + } +} + +static int clamp_int(int v, int lo, int hi) +{ + return v < lo ? lo : (v > hi ? hi : v); +} + +static void fetch_yuv(const struct xpsb_yuv_src *s, int x, int y, int *yuv) +{ + const uint8_t *p; + + x = clamp_int(x, 0, (int)s->w - 1); + y = clamp_int(y, 0, (int)s->h - 1); + + switch (s->fourcc) { + case XPSB_FOURCC_YUY2: + p = s->plane[0] + (ptrdiff_t)y * s->stride[0] + (x & ~1) * 2; + yuv[0] = p[(x & 1) * 2]; + yuv[1] = p[1]; + yuv[2] = p[3]; + return; + case XPSB_FOURCC_UYVY: + p = s->plane[0] + (ptrdiff_t)y * s->stride[0] + (x & ~1) * 2; + yuv[0] = p[1 + (x & 1) * 2]; + yuv[1] = p[0]; + yuv[2] = p[2]; + return; + case XPSB_FOURCC_NV12: + yuv[0] = s->plane[0][(ptrdiff_t)y * s->stride[0] + x]; + p = s->plane[1] + (ptrdiff_t)(y >> 1) * s->stride[1] + + (x >> 1) * 2; + yuv[1] = p[0]; + yuv[2] = p[1]; + return; + default: + /* YV12 and I420 both reach us as Y, U, V; psbCopyPlanarYUVData + * normalises the plane order when it fills the buffer. */ + yuv[0] = s->plane[0][(ptrdiff_t)y * s->stride[0] + x]; + yuv[1] = s->plane[1][(ptrdiff_t)(y >> 1) * s->stride[1] + + (x >> 1)]; + yuv[2] = s->plane[2][(ptrdiff_t)(y >> 1) * s->stride[2] + + (x >> 1)]; + return; + } +} + +static void sample_yuv(const struct xpsb_yuv_src *s, float fx, float fy, + int linear, int *yuv) +{ + int x, y, wx, wy, i; + int a[3], b[3], c[3], d[3]; + + if (!linear) { + fetch_yuv(s, (int)floorf(fx), (int)floorf(fy), yuv); + return; + } + + fx -= 0.5f; + fy -= 0.5f; + x = (int)floorf(fx); + y = (int)floorf(fy); + wx = (int)((fx - (float)x) * 256.0f + 0.5f); + wy = (int)((fy - (float)y) * 256.0f + 0.5f); + + fetch_yuv(s, x, y, a); + fetch_yuv(s, x + 1, y, b); + fetch_yuv(s, x, y + 1, c); + fetch_yuv(s, x + 1, y + 1, d); + + for (i = 0; i < 3; i++) { + int top = a[i] + (((b[i] - a[i]) * wx + 128) >> 8); + int bot = c[i] + (((d[i] - c[i]) * wx + 128) >> 8); + + yuv[i] = top + (((bot - top) * wy + 128) >> 8); + } +} + +void xpsb_yuv_blit(const struct xpsb_image *dst, int dx, int dy, + int dw, int dh, float u0, float v0, float u1, float v1, + const struct xpsb_yuv_src *src, + const struct xpsb_yuv_conv *conv) +{ + unsigned int dbytes = dst->fmt.bpp >> 3; + int linear, i, j; + float sx_span, sy_span, x_base, y_base, dfx, dfy; + + if (dw <= 0 || dh <= 0 || !xpsb_yuv_planes(src->fourcc)) + return; + + /* Point sampling is both faster and exact at 1:1, which is the common + * case for an unscaled overlay. */ + sx_span = (u1 - u0) * (float)src->w; + sy_span = (v1 - v0) * (float)src->h; + linear = src->filter == XPSB_FILTER_LINEAR && + (fabsf(sx_span - (float)dw) > 0.01f || + fabsf(sy_span - (float)dh) > 0.01f); + + x_base = u0 * (float)src->w; + y_base = v0 * (float)src->h; + dfx = sx_span / (float)dw; + dfy = sy_span / (float)dh; + + for (j = 0; j < dh; j++) { + int y = dy + j; + float fy = y_base + ((float)j + 0.5f) * dfy; + uint8_t *row; + + if (y < 0 || y >= (int)dst->h) + continue; + row = dst->base + (ptrdiff_t)y * dst->stride; + + for (i = 0; i < dw; i++) { + int x = dx + i; + float fx = x_base + ((float)i + 0.5f) * dfx; + int yuv[3], k; + uint32_t argb = 0xff000000u; + + if (x < 0 || x >= (int)dst->w) + continue; + sample_yuv(src, fx, fy, linear, yuv); + + for (k = 0; k < 3; k++) { + int32_t v = conv->c[k][0] * yuv[0] + + conv->c[k][1] * yuv[1] + + conv->c[k][2] * yuv[2] + conv->k[k]; + + v >>= conv->shift[k]; + argb |= (uint32_t)clamp_int((int)v, 0, 255) + << (16 - 8 * k); + } + xpsb_pix_store(&dst->fmt, + row + (ptrdiff_t)x * dbytes, argb); + } + } +} diff -urNp mesa-26.2.2.orig/src/gallium/drivers/sgx/xpsb_yuv.h mesa-26.2.2/src/gallium/drivers/sgx/xpsb_yuv.h --- mesa-26.2.2.orig/src/gallium/drivers/sgx/xpsb_yuv.h 1970-01-01 01:00:00.000000000 +0100 +++ mesa-26.2.2/src/gallium/drivers/sgx/xpsb_yuv.h 2026-09-08 10:57:36.685062405 +0200 @@ -0,0 +1,52 @@ +/* Open replacement for Xpsb.so - YUV to RGB video blit. + * + * Copyright (C) 2026 Rene Rebe + * + * SPDX-License-Identifier: MIT + */ +#ifndef _XPSB_YUV_H_ +#define _XPSB_YUV_H_ + +#include "xpsb_pixel.h" + +#define XPSB_FOURCC(a, b, c, d) ((uint32_t)(a) | ((uint32_t)(b) << 8) | \ + ((uint32_t)(c) << 16) | ((uint32_t)(d) << 24)) +#define XPSB_FOURCC_YUY2 XPSB_FOURCC('Y', 'U', 'Y', '2') +#define XPSB_FOURCC_UYVY XPSB_FOURCC('U', 'Y', 'V', 'Y') +#define XPSB_FOURCC_YV12 XPSB_FOURCC('Y', 'V', '1', '2') +#define XPSB_FOURCC_I420 XPSB_FOURCC('I', '4', '2', '0') +#define XPSB_FOURCC_NV12 XPSB_FOURCC('N', 'V', '1', '2') + +/* Integer form of the colour space conversion, one row per RGB component: + * out = (c[i][0] * Y + c[i][1] * U + c[i][2] * V + k[i]) >> shift[i]. */ +struct xpsb_yuv_conv { + int32_t c[3][3]; + int32_t k[3]; + int shift[3]; +}; + +struct xpsb_yuv_src { + const uint8_t *plane[3]; + int stride[3]; + unsigned int w, h; /* luma dimensions in pixels */ + uint32_t fourcc; + int filter; +}; + +/* The DDX hands packed formats a float matrix already divided by the luma + * range, with the luma offset in element 9; planar formats get the nine + * packed fixed-point words psb_pack_coeffs builds. */ +void xpsb_yuv_conv_packed(const float *conversion_data, + struct xpsb_yuv_conv *conv); +void xpsb_yuv_conv_planar(const uint32_t *sgx_coeffs, + struct xpsb_yuv_conv *conv); +void xpsb_yuv_conv_default(struct xpsb_yuv_conv *conv); + +int xpsb_yuv_planes(uint32_t fourcc); + +void xpsb_yuv_blit(const struct xpsb_image *dst, int dx, int dy, + int dw, int dh, float u0, float v0, float u1, float v1, + const struct xpsb_yuv_src *src, + const struct xpsb_yuv_conv *conv); + +#endif /* _XPSB_YUV_H_ */ diff -urNp mesa-26.2.2.orig/src/gallium/meson.build mesa-26.2.2/src/gallium/meson.build --- mesa-26.2.2.orig/src/gallium/meson.build 2026-09-02 17:40:02.000000000 +0200 +++ mesa-26.2.2/src/gallium/meson.build 2026-09-08 10:57:36.694204917 +0200 @@ -160,6 +160,11 @@ if with_gallium_i915 else driver_i915 = declare_dependency() endif +if with_gallium_sgx + subdir('drivers/sgx') +else + driver_sgx = declare_dependency() +endif if with_gallium_svga if not with_platform_windows subdir('winsys/svga/drm') diff -urNp mesa-26.2.2.orig/src/gallium/targets/dri/meson.build mesa-26.2.2/src/gallium/targets/dri/meson.build --- mesa-26.2.2.orig/src/gallium/targets/dri/meson.build 2026-09-02 17:40:02.000000000 +0200 +++ mesa-26.2.2/src/gallium/targets/dri/meson.build 2026-09-08 10:57:36.694424427 +0200 @@ -59,7 +59,7 @@ libgallium_dri = shared_library( dep_libdrm, dep_llvm, dep_thread, idep_xmlconfig, idep_mesautil, driver_swrast, driver_r300, driver_r600, driver_radeonsi, driver_nouveau, driver_kmsro, driver_v3d, driver_vc4, driver_freedreno, driver_etnaviv, - driver_tegra, driver_i915, driver_svga, driver_virgl, + driver_tegra, driver_i915, driver_sgx, driver_svga, driver_virgl, driver_panfrost, driver_iris, driver_lima, driver_zink, driver_d3d12, driver_asahi, driver_crocus, driver_rocket, driver_ethosu ], diff -urNp mesa-26.2.2.orig/src/gallium/targets/dril/dril_target.c mesa-26.2.2/src/gallium/targets/dril/dril_target.c --- mesa-26.2.2.orig/src/gallium/targets/dril/dril_target.c 2026-09-02 17:40:02.000000000 +0200 +++ mesa-26.2.2/src/gallium/targets/dril/dril_target.c 2026-09-08 10:57:36.694405057 +0200 @@ -631,6 +631,8 @@ PUBLIC const __DRIextension **__driDrive } DEFINE_LOADER_DRM_ENTRYPOINT(i915) +DEFINE_LOADER_DRM_ENTRYPOINT(gma500) +DEFINE_LOADER_DRM_ENTRYPOINT(sgx) DEFINE_LOADER_DRM_ENTRYPOINT(iris) DEFINE_LOADER_DRM_ENTRYPOINT(crocus) DEFINE_LOADER_DRM_ENTRYPOINT(nouveau)