diff --git a/src/core/gpu_hw_shadergen.cpp b/src/core/gpu_hw_shadergen.cpp index 33d9819f3..597beac95 100644 --- a/src/core/gpu_hw_shadergen.cpp +++ b/src/core/gpu_hw_shadergen.cpp @@ -835,251 +835,951 @@ void FilteredSampleFromVRAM(TEXPAGE_VALUE texpage, float2 coords, float4 uv_limi } else if (texture_filter == GPUTextureFilter::MMPXEnhanced) { - ss << "#define src(xoffs, yoffs) packUnorm4x8(SampleFromVRAM(texpage, bcoords + float2((xoffs), (yoffs)), " - "uv_limits))\n"; - - /* - * This part of the shader is from MMPX.glc from https://casual-effects.com/research/McGuire2021PixelArt/index.html - * Copyright 2020 Morgan McGuire & Mara Gagiu. - * Provided under the Open Source MIT license https://opensource.org/licenses/MIT + ss << R"( + #define srcf(xoffs,yoffs) SampleFromVRAM(texpage, bcoords + float2((xoffs), (yoffs)), uv_limits) + #define src(xoffs,yoffs) packUnorm4x8(srcf(xoffs,yoffs)) + )"; + + /* MMPXEnhanced v3.0 + * This shader is an enhanced iteration of MMPX.glc (see above). + * It improves the visual quality while preserving the pixel-art aesthetic by + * identifying and analyzing specific geometric shapes, effectively resolving + * the artifacts found in the original algorithm. + * + * (c) 2025-2026 by crashGG. + * Licensed under the same terms as MMPX.glc. */ + ss << R"( -uint luma(uint C) { - uint alpha = (C & 0xFF000000u) >> 24; - return (((C & 0x00FF0000u) >> 16) + ((C & 0x0000FF00u) >> 8) + (C & 0x000000FFu) + 1u) * (256u - alpha); + +//RGB visual weight + alpha segmentation +float luma(float4 col) { + + //Use CRT-era BT.601 standard. Clamp range to [0.0 - 0.999] + float rgbsum =min(dot(col.rgb, float3(0.299, 0.587, 0.114)), 0.9999); + + // Alpha weighting can be removed for subsequent fractional bit extraction + float alphafactor = + (col.a > 0.998) ? 0.0 : // Opaque + (col.a > 0.5) ? 2.0 : + (col.a > 0.002) ? 4.0 : 6.0; // Fully transparent + + return rgbsum + alphafactor; } -bool all_eq2(uint B, uint A0, uint A1) { - return ((B ^ A0) | (B ^ A1)) == 0u; +/* Constant explanations: +0.145898 : Double short golden ratio of 1.0 +0.0638587 : Squared double short golden ratio of RGB Euclidean distance +0.4377 : Squared single short golden ratio of RGB Euclidean distance +0.75 : Squared half of RGB Euclidean distance +*/ + +// duck calculation: dot(diff,diff) directly incorporates alpha channel //duck.alpha +// Note: Transparent duck pixels are determined outside sim and mixFactor functions +bool simb(float4 col1, float4 col2) { + + float4 diff = col1 - col2; + + float maxdiff = max(diff.r, max(diff.g, diff.b)); + float mindiff = min(diff.r, min(diff.g, diff.b)); + + // Luminance baseline weight: both colors must be > 0.078 (0.234/3) + float weight = step(0.234, min(col1.r+col1.g+col1.b, col2.r+col2.g+col2.b)); + + // Find the most opposite channel: if one positive and one negative, take the smallest absolute value; 0 for same direction + // Use max(0.0, ...) to filter same-sign cases + // Skip team_rebel if either pixel luminance < 0.078 + float team_rebel = min(max(0.0, maxdiff), max(0.0, -mindiff)) * weight; + float finaldist = (maxdiff - mindiff) + team_rebel; + + float dot_diff = dot(diff, diff); + + // Equivalent to (finaldist / 0.145898 )^2 + float factor = (finaldist * finaldist) * 46.9787; + + return dot_diff < mix(0.0638587, 0.0, factor); } -bool all_eq3(uint B, uint A0, uint A1, uint A2) { - return ((B ^ A0) | (B ^ A1) | (B ^ A2)) == 0u; +bool sim(float4 col1, float4 col2) { + + float4 diff = col1 - col2; + + // RGB color difference range (max_diff - min_diff) + float delta_range = max(diff.r, max(diff.g, diff.b)) - min(diff.r, min(diff.g, diff.b)); + + float dot_diff = dot(diff, diff); + + // Equivalent to (delta_range / 0.382 )^2 + float factor = (delta_range * delta_range) * 6.8541; + + return dot_diff < mix(0.0638587, 0.0, factor); } -bool all_eq4(uint B, uint A0, uint A1, uint A2, uint A3) { - return ((B ^ A0) | (B ^ A1) | (B ^ A2) | (B ^ A3)) == 0u; +bool vi_sim(float4 col1, uint uC1, uint uC2) { + if (uC1==uC2) return true; + float4 col2 = unpackUnorm4x8(uC2); // duck.alpha + return sim(col1, col2); } -bool any_eq3(uint B, uint A0, uint A1, uint A2) { - return B == A0 || B == A1 || B == A2; +float mixGate(float4 col1, float4 col2) { + + float4 diff = col1 - col2; + + // RGB color difference range (max_diff - min_diff) + float delta_range = max(diff.r, max(diff.g, diff.b)) - min(diff.r, min(diff.g, diff.b)); + + float dot_diff = dot(diff, diff); + + // Equivalent to (delta_range / 0.618 )^2 + float factor = (delta_range * delta_range) * 2.618034; + + return step(dot_diff, mix(0.75, 0.0, factor)); } -bool none_eq2(uint B, uint A0, uint A1) { - return (B != A0) && (B != A1); + +#define eq(a,b) (a==b) + +#define neq(a,b) (a!=b) + +#define all_eq2(a, b1, b2) \ + ( eq(a,b1) && eq(a,b2)) + +#define all_eq3(a, b1, b2, b3) \ + ( eq(a,b1) && eq(a,b2) && eq(a,b3)) + +#define all_eq4(a, b1, b2, b3, b4) \ + ( eq(a,b1) && eq(a,b2) && eq(a,b3) && eq(a,b4)) + +#define any_eq2(a, b1, b2) (eq(a,b1)||eq(a,b2)) +#define any_eq3(a, b1, b2, b3) (eq(a,b1)||eq(a,b2)||eq(a,b3)) +// Better than a!=b1 && a!=b2 +#define none_eq2(a, b1, b2) !any_eq2(a, b1, b2) + + +// Pre-define +//#define testcolor float4(1.0, 0.0, 1.0, 1.0) // Magenta +//#define testcolor2 float4(0.0, 1.0, 1.0, 1.0) // Cyan +//#define testcolor3 float4(1.0, 1.0, 0.0, 1.0) // Yellow +//#define testcolor4 float4(1.0, 1.0, 1.0, 1.0) // White +#define slopOFF float4(2.0, 2.0, 2.0, 2.0) +#define slopeBAD float4(4.0, 4.0, 4.0, 4.0) +#define theEXIT float4(8.0, 8.0, 8.0, 8.0) + +#define mixXE mix(vX,vE,mixFactor) +#define mixXEoff mixXE+slopOFF +#define Xoff vX+slopOFF +//#define checkblack(col) ((col).g < 0.078 && (col).r < 0.1 && (col).b < 0.1) + +#if API_OPENGL || API_OPENGL_ES || API_VULKAN + + #define checkblack(col) all(lessThan((col).rgb, float3(0.1, 0.078, 0.1))) + #define checkwhite(col) all(greaterThan((col).rgb, float3(0.92, 0.92, 0.92))) + #define vec_neq(a, b) any(greaterThan(abs((a)-(b)), float4(0.01))) + +#else + + #define checkblack(col) all((col).rgb < float3(0.1, 0.078, 0.1)) + #define checkwhite(col) all((col).rgb > float3(0.92, 0.92, 0.92)) + #define vec_neq(a, b) any(abs((a)-(b)) > 0.01) + +#endif +)" + R"( +//pin zz +// "Concave + Cross" type weak blending (weak blend / none) +float4 admixC(float4 vX, float4 vE) { + // Weak blending. Blend enabled? 0.618 else 1.0 + float mixFactor = mixGate(vX, vE) * (-0.381966) + 1.0; + + return mixXE; } -bool none_eq4(uint B, uint A0, uint A1, uint A2, uint A3) { - return B != A0 && B != A1 && B != A2 && B != A3; +// K-type forced weak blending +float4 admixK(float4 vX, float4 vE) { + float4 diff = vX - vE; + // mixFactor slides from 0.5-1.0 based on point set distance, quadratic curve, steeper closer to 1.0 + float mixFactor = dot(diff.rgb, diff.rgb) * 0.16666 + 0.5; // xxx.alpha + // mixFactor slides linearly from 0.5-1.0 based on Euclidean distance + //float mixFactor = distance(vX, vE) * 0.28867 + 0.5; + return mixXE; } +// L-type 2:1 slope Main corner extension +// Practice: This rule requires 4 pixels on strict slope to be identical. Otherwise various artifacts will appear! +float4 admixL(float4 vX, float4 vE, float4 vS) { -// Two-stage weak blending, mix/none -uint admix2d(uint a, uint b) { - float4 a_float = unpackUnorm4x8(a); - float4 b_float = unpackUnorm4x8(b); - float3 diff_rgb = a_float.rgb - b_float.rgb; - float rgbDist = dot(diff_rgb, diff_rgb); - - // Combine conditional judgments (reduce branches) - bool aIsBlack = dot(a_float.rgb, a_float.rgb) < 0.01; - //bool aIsTransparent = a_float.a < 0.01; - //bool bIsTransparent = b_float.a < 0.01; - - if (aIsBlack ) return b; - - // Determine blending mode based on distance - float4 result; - if (rgbDist < 1.0) { - // Close distance: linearly blend RGB and Alpha - result = (a_float + b_float) * 0.5; - } else { - // Far distance: return b - result = b_float; - } - - // Repack as uint - return packUnorm4x8(result); -} - - -/*============================================================================= -Auxiliary function for 4-pixel cross determination: scores the number of matches at specific positions of the pattern. -Three pattern conditions are determined, requiring 6 points to be satisfied. - ┌───┬───┬───┐ ┌───┬───┬───┐ - │ A │ B │ C │ │ A │ B │ 1 │ - ├───┼───┼───┤ ├───┼───┼───┤ - │ D │ E │ F │ => L │ B │ A │ 2 │ - ├───┼───┼───┤ ├───┼───┼───┤ - │ G │ H │ I │ │ 5 │ 4 │ 3 │ - └───┴───┴───┘ └───┴───┴───┘ -=============================================================================*/ - -bool countPatternMatches(uint LA, uint LB, uint L1, uint L2, uint L3, uint L4, uint L5) { - - int score1 = 0; // Diagonal pattern 1 - int score2 = 0; // Diagonal pattern 2 - int score3 = 0; // Horizontal/vertical line pattern - int scoreBonus = 0; - - // Replace Euclidean formula with dot product to save a square root calculation - float4 a_float = unpackUnorm4x8(LA); - float4 b_float = unpackUnorm4x8(LB); - float3 diff_rgb = a_float.rgb - b_float.rgb; - float rgbDist = dot(diff_rgb, diff_rgb); + // Check eqX,E: Originally captured many duplicate pixels, now main thread passes slopeok filter. + + // If target X and reference S(sample) differ, blending has occurred once; direct copy, no re-blending + if (vec_neq(vX, vS)) return vX; + + float mixFactor = 0.381966 * mixGate(vX,vE) * step(0.002,vE.r+vE.g+vE.b+vE.a); // xxx.alpha.duck + + return mixXE; +} +float4 admixX( uint A, uint B, uint C, uint D, uint E, uint F, uint G, uint H, uint I + , uint P, uint PA, uint PC, uint Q, uint QA, uint QG, uint R, uint RC, uint RI, uint S, uint SG, uint SI, uint AA, uint CC, uint GG + , float El, float Bl, float Dl, float Fl, float Hl + , float4 vE, float4 vB, float4 vD, float4 vC, float4 vG + ) { + + + bool eq_B_C = eq(B,C); + bool eq_D_G = eq(D,G); + + if (eq_B_C && eq_D_G) return slopeBAD; + + + //Pre-declare + bool eq_B_P; bool eq_B_PA; bool eq_B_PC; + bool eq_D_Q; bool eq_D_QA; bool eq_D_QG; + bool eq_E_F; bool eq_E_H; bool eq_A_AA; + + float4 vX; + float mixFactor; + + bool eq_E_C = eq(E,C); + bool eq_E_G = eq(E,G); + bool eq_A_P = eq(A,P); + bool eq_A_Q = eq(A,Q); + bool comboE3 = eq_E_C && eq_E_G; + bool comboA3 = eq_A_P && eq_A_Q; + +if (neq(B,D)){ + + if (eq(E,A)) return slopeBAD; + + float diffBD = abs(Bl-Dl); + if (diffBD > El-Bl || diffBD > El-Dl) return slopeBAD; + + + vX = mix(vB, vD, 0.5); + vX.a = min(vB.a, vD.a); + + mixFactor = 0.381966 * mixGate(vX,vE) * float(E!=0u); + + eq_B_PC = eq(B,PC); + eq_D_QG = eq(D,QG); - // Add details for very close colors, reduce details for highly different colors (font edges) - if (rgbDist < 0.06386) { // Point set after quadratic golden section, colors are quite close - scoreBonus += 1; - } else if (rgbDist > 2.18847) { // Point set after quadratic golden section, significant difference - scoreBonus -= 1; - } - - // Diagonals use a deduction system: deduct points for crosses, add back if conditions are met - // 1. Diagonal pattern ╲ (Condition: B = 2 or 4) - if (LB == L2 || LB == L4) { - score1 -= int(LB == L2 && LA == L1) * 1; // A-1 and B-2 form a cross, deduct points - score1 -= int(LB == L4 && LA == L5) * 1; // A-5 and B-4 form a cross, deduct points - - // If the following triangular pattern is satisfied, offset the above cross deductions - score1 += int(LB == L1 && L1 == L2) * 1; // B-1-2 form a triangular pattern, add points - score1 += int(LB == L4 && L4 == L5) * 1; // B-4-5 form a triangular pattern, add points - score1 += int(L2 == L3 && L3 == L4) * 1; // 2-3-4 form a triangular pattern, add points - - score1 += scoreBonus + 6; - } - - // 2. Diagonal pattern ╱ (Condition: A = 1 or 5) - if (LA == L1 || LA == L5) { - score2 -= int(LB == L2 && LA == L1) * 1; // A-1 and B-2 form a cross, deduct points - score2 -= int(LB == L4 && LA == L5) * 1; // A-5 and B-4 form a cross, deduct points - score2 -= int(LA == L3) * 1; // A-3 forms a cross, deduct points - - // If the following triangular pattern is satisfied, offset the above cross deductions - score2 += int(LB == L1 && L1 == L2) * 1; // B-1-2 form a triangular pattern, add points - score2 += int(LB == L4 && L4 == L5) * 1; // B-4-5 form a triangular pattern, add points - score2 += int(L2 == L3 && L3 == L4) * 1; // 2-3-4 form a triangular pattern, add points - - score2 += scoreBonus + 6; - } - - // 3. Horizontal/vertical line pattern (Condition: horizontal continuity) uses a point addition system, passes only if conditions are met - if (LA == L2 || LB == L1 || LA == L4 || LB == L5 || (L1 == L2 && L2 == L3) || (L3 == L4 && L4 == L5)) { - score3 += int(LA == L2); // A equals 2, +1 - score3 += int(LB == L1); // B equals 1, +1 - score3 += int(L3 == L4); // 3 equals 4, +1 - score3 += int(L4 == L5); // 4 equals 5, +1 - score3 += int(L3 == L4 && L4 == L5); // 3-4-5 continuous - - score3 += int(LB == L5); // B equals 5, +1 - score3 += int(LA == L4); // A equals 4, +1 - score3 += int(L2 == L3); // 2 equals 3, +1 - score3 += int(L1 == L2); // 1 equals 2, +1 - score3 += int(L1 == L2 && L2 == L3); // 1-2-3 continuous - - // A x 4 square - score3 += int(LA == L2 && L2 == L3 && L3 == L4) * 2; - - // Patch for the previous rule to avoid bubbles in large cross patterns. Some games use single-side patterns, - // so it's best to expand for bilateral judgment (Work in Progress) - score3 -= int(LB == L1 && L1 == L5 && LA == L2 && L2 == L4)*3; - - score3 -= int(LA == L1 && LA == L5); // Deduct points if both L1 and L5 are A to avoid excessive scores - // and prevent the pattern from becoming a diagonal pattern. - - // Extra points - score3 += scoreBonus; // Experience: Even with very close colors, do not add too many points, - // as some Z-shaped crosses may produce bubbles. - } - - // Take the maximum of the four scores - int score = max(max(score1, score2), score3); - - return score < 6; // Requires 6 points to be satisfied + if (none_eq2(A,B,D)){ + if (comboA3) return mixXEoff; + if ( eq_A_P && eq_B_PC && !eq_B_C ) return mixXEoff; + if ( eq_A_Q && eq_D_QG && !eq_D_G ) return mixXEoff; + + if ( eq_A_P && eq_E_G ) return mixXEoff; + if ( eq_A_Q && eq_E_C ) return mixXEoff; + + if ( eq_E_C && eq_D_G ) return mixXEoff; + if ( eq_E_G && eq_B_C ) return mixXEoff; + +} + if ( comboE3 ) return mixXEoff; + + if ( eq_E_C && eq_B_PC && neq(B,P)) return mixXEoff; + if ( eq_E_G && eq_D_QG && neq(D,Q)) return mixXEoff; + + eq_E_F = eq(E,F); + + if (eq(F,H)) { + + if ( eq_E_C && !eq_D_G && (!eq_E_F||neq(E,P)) ) return mixXEoff; + if ( eq_E_G && !eq_B_C && (!eq_E_F||neq(E,Q)) ) return mixXEoff; + + if ( !eq_E_F && eq_B_PC && eq(F,RC) ) return mixXEoff; + if ( !eq_E_F && eq_D_QG && eq(H,SG) ) return mixXEoff; + } + + return slopeBAD; } + Bl = fract(Bl); + Dl = fract(Dl); + El = fract(El); + Fl = fract(Fl); + Hl = fract(Hl); + + bool Xisblack = checkblack(vB); + if ( Xisblack && El >0.5 && (Fl<0.078 || Hl<0.078) ) return theEXIT; + + vX = vB; + + mixFactor = 0.381966 * mixGate(vX,vE) * float(E!=0u); + + bool B_slope; bool B_tower; bool B_wall; + bool D_slope; bool D_tower; bool D_wall; + bool En3; + #define En4square En3&&eq(E,I) + +if (eq(E,A)) { + + eq_E_F = eq(E,F); + eq_E_H = eq(E,H); + + bool Eisblack = checkblack(vE); + + if ( comboE3 && !eq_E_F && !eq_E_H && eq(E,I) ) { + + if (Eisblack) return theEXIT; + mixFactor = 0.618034 * (1.0 - mixFactor); + return mixXEoff; + } + + eq_A_AA = eq(A,AA); + + if ( comboA3 && eq_A_AA && none_eq2(A,PA,QA) ) { + if (Eisblack) return theEXIT; + mixFactor = 0.618034 * (1.0 - mixFactor); + if ( neq(B,PA) && eq(PA,QA) ) return mixXEoff; + mixFactor += 0.236068; + return mixXEoff; + } + + eq_B_PC = eq(B,PC); + eq_B_PA = eq(B,PA); + eq_D_QG = eq(D,QG); + eq_D_QA = eq(D,QA); + + if ( comboE3 && comboA3 && + (eq_B_PC || eq_D_QG) && eq_D_QA && eq_B_PA) { + mixFactor = mixFactor * (-0.618034) + 0.8541; + return mixXEoff; + } + + if ( comboE3 && eq_A_P + && eq_B_PA && eq_D_QA && eq_D_QG + && eq_E_H + ) { + mixFactor = mixFactor * (-0.618034) + 0.8541; + return mixXEoff; + } + + if ( comboE3 && eq_A_Q + && eq_B_PA && eq_D_QA && eq_B_PC + && eq_E_F + ) { + mixFactor = mixFactor * (-0.618034) + 0.8541; + return mixXEoff; + } + + if (comboA3) return Xoff; + + if (comboE3) return mixXEoff; + + eq_B_P = eq(B, P); + eq_D_Q = eq(D, Q); + + B_slope = eq_B_PC && !eq_B_P && !eq_B_C && !eq_B_PA; + D_slope = eq_D_QG && !eq_D_Q && !eq_D_G && !eq_D_QA; + + B_wall = eq_B_C && !eq_B_PC && !eq_B_P; + D_wall = eq_D_G && !eq_D_QG && !eq_D_Q; + + B_tower = eq_B_P && !eq_B_PC && !eq_B_C && !eq_B_PA; + D_tower = eq_D_Q && !eq_D_QG && !eq_D_G && !eq_D_QA; + + if ( B_slope && eq_E_G ) return mixXEoff; + if ( D_slope && eq_E_C ) return mixXEoff; + + float scoreE = 0.0; + float scoreB = 0.0; + float scoreD = 0.0; + float scoreZ = 0.0; + + if (eq_E_C) { + scoreE += 1.0 +float(eq(F,H)) +float(B_slope); + scoreE -= float(all_eq2(E,P,PC)&&!D_wall); + } + + if (eq_E_G) { + scoreE += 1.0 +float(eq(F,H)) +float(D_slope); + scoreE -= float(all_eq2(E,Q,QG)&&!B_wall); + } + + scoreE += float(B_slope && eq_A_Q || D_slope && eq_A_P); + + En3 = eq_E_F && eq_E_H; + + if ( scoreE<0.1 && mixFactor<0.1 && En4square && eq(E,S)==eq(E,SI) && eq(E,R)==eq(E,RI) ) return theEXIT; + + if ( scoreE<0.1 && !En3 && neq(E,I) ) { + if ( B_wall && eq_E_F ) return theEXIT; + if ( D_wall && eq_E_H ) return theEXIT; + } + + scoreE += float(B_slope && eq_A_P || D_slope && eq_A_Q); + + if ( !En3 && eq(F,H) ) { + if (Eisblack) return slopeBAD; + bool condZ1 = B_wall && (eq(F,R) || eq(F,RC) || eq(G,H) || eq(F,I)); + bool condZ2 = D_wall && (eq(C,F) || eq(H,SG) || eq(H,S) || eq(F,I)); + scoreZ = float(condZ1 || condZ2); + } + + if (eq_B_PA) { + scoreB -= 1.0 +float(eq(P,C)) +float(eq_A_AA); + } + + if (eq(P,C)){ + scoreB -= float(eq_A_AA); + scoreZ *= float(scoreE < 0.1); + } + + if (eq_D_QA) { + scoreD -= 1.0 +float(eq(G,Q)) +float(eq_A_AA); + } + + if (eq(G,Q)){ + scoreD -= float(eq_A_AA); + scoreZ *= float(scoreE < 0.1); + } + + float scoreFinal = scoreE + scoreB + scoreD + scoreZ ; + + scoreFinal += float(min(scoreB,scoreD) > -0.1 && (B_wall && D_tower || B_tower && D_wall)) *2.0; + + mixFactor *= (1.0 - step(1.9, scoreFinal)); + return mixXE + slopeBAD*(1.0 - step(0.9, scoreFinal)); + +} + + if (eq_E_C ) { + if (comboA3) return vX; + if (comboE3) return mixXE; + if (all_eq2(B,A,PA) && all_eq3(E,F,P,PC)) return theEXIT; + return mixXE; + } + + if (eq_E_G) { + if (comboA3) return vX; + if (comboE3) return mixXE; + if (all_eq2(D,A,QA) && all_eq3(E,H,Q,QG)) return theEXIT; + return mixXE; + } + + if (E==0u) return theEXIT; // xxx.alpha + + bool eq_A_B = eq(A,B); + bool eq_F_H = eq(F,H); + + eq_B_P = eq(B,P); + eq_B_PC = eq(B,PC); + eq_B_PA = eq(B,PA); + eq_D_Q = eq(D,Q); + eq_D_QG = eq(D,QG); + eq_D_QA = eq(D,QA); + + B_slope = eq_B_PC && !eq_B_P && !eq_B_C; + D_slope = eq_D_QG && !eq_D_Q && !eq_D_G; + B_tower = eq_B_P && !eq_B_PC && !eq_B_C && !eq_B_PA; + D_tower = eq_D_Q && !eq_D_QG && !eq_D_G && !eq_D_QA; + B_wall = eq_B_C && !eq_B_PC && !eq_B_P; + D_wall = eq_D_G && !eq_D_QG && !eq_D_Q; + + if (!eq_A_B) { + + if (comboA3) return Xoff; + + if ( (B_slope||B_tower) && (D_slope||D_tower) ) return Xoff; + + if ( B_slope && eq_A_P ) return mixXEoff; + if ( D_slope && eq_A_Q ) return mixXEoff; + + if ( (B_slope || D_slope) && eq_F_H ) return mixXEoff; + + if ( B_slope && eq(H,SG) ) return mixXEoff; + if ( D_slope && eq(F,RC) ) return mixXEoff; + + if ( B_slope && eq_A_Q && eq(Q,QG) ) return mixXEoff; + if ( D_slope && eq_A_P && eq(P,PC) ) return mixXEoff; + + } + + bool sim_EC = E!=0u && C!=0u && sim(vE, vC); + bool sim_EG = E!=0u && G!=0u && sim(vE, vG); + + float E_lumDiff = mix(0.381966, 0.145898, max((El - 0.8541),0.0) * 6.8541); + + if ( mixFactor<0.1 && !sim_EC && !sim_EG && neq(E,I) && abs(El-Fl)>E_lumDiff && abs(El-Hl)>E_lumDiff ) return slopeBAD; + + eq_E_F = eq(E,F); + eq_E_H = eq(E,H); + + if ( eq_B_C && eq_D_Q ) { + if ( eq(P,PC) && eq(A,QA) && !eq_D_QG && eq_E_F && !eq_E_H && eq(H,I)) return theEXIT; + if ( eq_A_B ) return slopeBAD; + if ( B_wall && D_tower && eq_E_F) return vX; + return mixXEoff; + } + + if ( eq(D,G) && eq(B,P)) { + if ( eq(Q,QG) && eq(A,PA) && !eq_B_PC && eq_E_H && !eq_E_F && eq(F,I)) return theEXIT; + if ( eq_A_B ) return slopeBAD; + if ( B_tower && D_wall && eq_E_H) return vX; + return mixXEoff; + } + + En3 = eq_E_F && eq_E_H; + + if ( En4square ) { + if ( ( eq_B_C || eq_D_G) && eq_A_B) return theEXIT; + if ( ( eq_B_C || eq_D_G || mixFactor<0.1) && (eq(E,S) == eq(E, SI) && eq(E,R) == eq(E, RI)) ) return theEXIT; + return mixXEoff; + } + + if (!eq_B_C && !eq_D_G ) { + if ( comboA3 && eq_F_H ) return Xoff; + + if ( comboA3&&eq_B_PC&&eq(C,CC) ) return Xoff; + if ( comboA3&&eq_D_QG&&eq(G,GG) ) return Xoff; + + if ( !eq_B_P && !eq_B_PC && !eq_D_Q && !eq_D_QG && !En3 ) return slopeBAD; + + if (eq_A_Q&&sim_EC) return mixXEoff; + if (eq_A_P&&sim_EG) return mixXEoff; + if (sim_EC&&sim_EG ) return mixXEoff; + } + + if ( En3 && eq_A_B) return theEXIT; + + if (eq_F_H) { + + if ( eq_B_PC&&eq(F,RC) || eq_D_QG&&eq(H,SG) ) return mixXEoff; + + if (eq_A_B) return slopeBAD; + + if ( eq_B_C || eq_D_G) return mixXEoff; + if ( eq_B_PC || eq_D_QG) return mixXEoff; + + } + + return slopeBAD; + +} + +float4 admixS( uint A, uint B, uint C, uint D, uint E, uint F, uint G, uint H, uint I + , uint R, uint RC, uint RI, uint S, uint SG, uint SI, uint II, uint CC + , float4 vE, float4 vF, float4 vC + ) { + + if (any_eq2(F,C,I)) return vE; + + if ( (eq(F,RI) || eq(G,S) || eq(R, RI)) && neq(R,I) ) return vE; + + if (eq(H, S) && none_eq2(H,I,SG)) return vE; + + if ( eq(R, RC) || eq(G,SG) ) return vE; + + if ( checkwhite(vE) && all_eq2(E,C,D) && none_eq2(E,RC,CC)) return vE; + + #define vX vF + float mixFactor = 0.381966 * mixGate(vX,vE) * float(E!=0); + + if ( eq(E,C) && (eq(E,D)||eq(B,D)) ) return mixXE; + + bool sim_E_C = E!=0u && E!=0u && sim(vE,vC); + + if ( sim_E_C && eq(E,D) && eq(B,C) ) return mixXE; + + if ( (sim_E_C || mixFactor>0.1) && all_eq2(B,C,D) ) return mixXE; + + return vE; +} +)" + R"( void FilteredSampleFromVRAM(TEXPAGE_VALUE texpage, float2 coords, float4 uv_limits, out float4 texcol, out float ialpha) { float2 bcoords = floor(coords); - uint A = src(-1, -1), B = src(+0, -1), C = src(+1, -1); - uint D = src(-1, +0), E = src(+0, +0), F = src(+1, +0); - uint G = src(-1, +1), H = src(+0, +1), I = src(+1, +1); + float4 vE = SampleFromVRAM(texpage, bcoords, uv_limits); + + float4 vB = srcf(0.0, -1.0); + float4 vD = srcf(-1.0, 0.0); + float4 vF = srcf(+1.0, 0.0); + float4 vH = srcf(0.0, +1.0); + + uint E = packUnorm4x8(vE); + uint B = packUnorm4x8(vB); + uint D = packUnorm4x8(vD); + uint F = packUnorm4x8(vF); + uint H = packUnorm4x8(vH); + + // default pixel + ialpha = float(E != 0u); + texcol = vE; + + bool eq_E_D = eq(E,D); + bool eq_E_F = eq(E,F); + bool eq_E_B = eq(E,B); + bool eq_E_H = eq(E,H); + bool eq_B_H = eq(B,H); + bool eq_D_F = eq(D,F); + + +bool skiprest = (eq_E_D && eq_E_F) || (eq_E_B && eq_E_H) || (eq_B_H && eq_D_F); +if (!skiprest) { + + + + // 5x5 + float4 vA = srcf(-1.0, -1.0); + float4 vC = srcf(+1.0, -1.0); + float4 vG = srcf(-1.0, +1.0); + float4 vI = srcf(+1.0, +1.0); + + uint A = packUnorm4x8(vA); + uint C = packUnorm4x8(vC); + uint G = packUnorm4x8(vG); + uint I = packUnorm4x8(vI); + + uint P = src( 0.0, -2.0); + uint Q = src(-2.0, 0.0); + uint R = src(+2.0, 0.0); + uint S = src( 0.0, +2.0); + + uint PA = src(-1.0, -2.0); + uint PC = src(+1.0, -2.0); + uint QA = src(-2.0, -1.0); + uint QG = src(-2.0, +1.0); // AA PA [P] PC CC + uint RC = src(+2.0, -1.0); // |--------------| + uint RI = src(+2.0, +1.0); // QA | A | B | C | RC + uint SG = src(-1.0, +2.0); // |----|----|----| + uint SI = src(+1.0, +2.0); // [Q] | D | E | F | [R] + uint AA = src(-2.0, -2.0); // |----|----|----| + uint CC = src(+2.0, -2.0); // QG | G | H | I | RI + uint GG = src(-2.0, +2.0); // |----|----|----| + uint II = src(+2.0, +2.0); // GG SG [S] SI II + + + float4 J = vE; float4 K = vE; float4 L = vE; float4 M = vE; + + + float Bl = luma(vB) + float(B==0u); + float Dl = luma(vD) + float(D==0u); + float El = luma(vE) + float(E==0u); + float Fl = luma(vF) + float(F==0u); + float Hl = luma(vH) + float(H==0u); + +// pre-cal + bool eq_B_D = eq(B,D); + bool eq_B_F = eq(B,F); + bool eq_D_H = eq(D,H); + bool eq_F_H = eq(F,H); + + + bool oppoPix = eq_B_H || eq_D_F; + + bool slope1 = false; bool slope2 = false; bool slope3 = false; bool slope4 = false; + + bool slope1ok = false; bool slope2ok = false; bool slope3ok = false; bool slope4ok = false; + + +// B - D + if ( (B!=0u && D!=0u) && // xxx.alpha + (!eq_E_B && !eq_E_D && !oppoPix) && (!eq_D_H && !eq_B_F) + && (eq(E,A) || El>=Dl&&El>=Bl) && ( (El