/*{{{ Pixel-Planes V support code*/ /* ********************************** File pxpl5sup.c Project pazpl5 Author p j atkin (c) DIVISION Ltd 1993 */ /*}}} */ /*{{{ on pxpl5 oddness*/ /* pxpl5 forces you to do 3 things you normally wouldnt do on a graphics system - edgeize, planarize and binitize primitives edgeizing involves turning a polygon into a set of edges, each edge described by an expression of the form f(x,y) = Ax + By + C, where conventionally a point is INSIDE the edge if f(x,y) > 0 at (x,y) planarization is pretty similar, and is used for Z-buffering, Gouraud-shading and texturing. Planarization involves computing a screen-space planar equation for a given variable - so to Z-buffer, Gouraud-shade and texture a triangle we need to compute Z=fz(x,y), r=fr(x,y), g=fg(x,y), b=fb(x,y), u=fu(x,y), v=fv(x,y) where each of fz, fr, fg, fb, fu, fv are cast as expression of the form f=Ax + By + C binitization is different, and stems from the original MIMDness of pxpl5 - rather than build a 640x512 array of pixel-processors, we use multiple arrays of 128x128 (or 64x128) and if the polygons scatter statistically well, we can get many times more performance for a given number of pixel-processors in order to do this, as a triangle is transformed to screen-space, we need to determine how many screen-space regions of 64x128 are overlapped by the triangle, and place the triangle into 'bins' associated with each region. binitization may kill me yet. */ /*}}} */ /*{{{ includes*/ #include #include #include #include "DMAengn.h" #include "pxpl5typ.h" /*}}} */ /*{{{ void edgeize ( float *eqn, float *p1, float *p2 )*/ void edgeize ( float *eqn, float *p1, float *p2 ) { /* ********************* takes 2 point p1 and p2 and computes the edge equation edge Ax + By + C, +ve inside, -ve outside the edge 4 cases - a) p0 b) p1 \ / \ / \ / p1 p0 c) p1 d) p0 \ / \ / \ / p0 p1 We need to ensure that in all cases we treat the edges identically, e.g a = c with flipped vertices / opcode, ditto b, d */ float A, B, C; /* however, 1st approximation - this will suffer rounding errors + DDA cracks */ eqn[0]=p1[Y] - p2[Y]; eqn[1]=p2[X] - p1[X]; eqn[2]=(p2[Y]*p1[X]) - (p2[X]*p1[Y]); } /*}}} */ /*{{{ on planarization*/ /* Examination of the equations for planarization, and the UNC rendering library, indicates some useful speedups for planarizing. A recurring term is the divisor for all 3 coefficients, termed C. This is independent of the planarized variable; it only varies with screen-space X and Y, so can be precomputed once per triangle and re-used for all planarized expressions ditto some recurring difference expressions (x1 - x2 etc.) So preplanarize precomputes the useful stuff into a structure for subsequent planarizing. Ideally of course we precompute this into a set of floating-point registers. Later. Or maybe right now - how many registers do I need to do this? rx23 \ rx31 > the recurring differences rx12 / rC the divisor for preplanarizing; now to planarize, strive to keep rv1y rv2y rv3y rv1x rv2x and rv3x in registers also so we keep 10 fp registers hanging around, so to planarize a variable we access memory 3 times, to load v1 v2 v3 (which are used many times) using up just 13 fp registers then 3 writes to eqn->A, eqn->B, eqn->C this should be very fast indeed on an XP The correct structure for the code is probably a dispatch vector of C functions associated with each type of triangle - e.g 24-bit Gouraud, 8-bit Gouraud + intrinsic + spec 8-bit Gouraud + intrinsic + spec + texture + MIP etc, the function calls an assembly stub which preplanarizes, then repeatedly calls an assembly-coded planarize as many times as needed Planarization looks like a cost of 25 ticks per planarized variable, so we can planarize Z, diffuse, spec in 1.5 uS (667k triangles/sec) Z, diffuse, spec, u, v, homo in 3uS (333k ) Z, diffuse, spec, u, v, homo, MIP in 3.5uS, or (286k ) it looks like we can edgize in about 20 ticks per edge, or 1.2 uS per triangle, so the edgize / planarize costs become Z, diffuse, spec in 2.7 uS (370k triangles/sec) Z, diffuse, spec, u, v, homo in 4.2 uS (238k ) Z, diffuse, spec, u, v, homo, MIP in 3.5uS, or (212k ) or the VWE benchmark of flat-shaded textured quads - Z, u, v, homo, MIP in 3.6uS or (278k ) SO we are in shape performance-wise for planarization. How is binitizing. Read on. */ /*}}} */ /*{{{ void preplanarize ( preplane *p, float *v1, float *v2, float *v3 )*/ void preplanarize ( preplane *p, float *v1, float *v2, float *v3 ) { float v1x=v1[X]; float v2x=v2[X]; float v3x=v3[X]; float v1y=v1[Y]; float v2y=v2[Y]; float v3y=v3[Y]; p->x23=v2x - v3x; p->x31=v3x - v1x; p->x12=v1x - v2x; p->C=1.0f / ((v1x * (v2x - v3x)) + (v2x * (v3x - v1x)) + (v3x * (v1x - v2x))); } /*}}} */ /*{{{ void planarize ( float *eqn, float *v1, float *v2, float *v3, int index, preplane *p )*/ void planarize ( float *eqn, float *v1, float *v2, float *v3, int index, preplane *p ) { float v1x=v1[X]; float v2x=v2[X]; float v3x=v3[X]; float v1y=v1[Y]; float v2y=v2[Y]; float v3y=v3[Y]; float v1f=v1[index]; float v2f=v2[index]; float v3f=v3[index]; eqn[0]=-p->C* ((v1y * (v2f - v3f)) + (v2y * (v3f - v1f)) + (v3y * (v1f - v2f))); eqn[1]=-p->C*((v1f * p->x23) + (v2f * p->x31) + (v3f * p->x12)); eqn[2]= p->C*((v1x*((v2y*v3f) - (v3y*v2f))) + (v2x*((v3y*v1f) - (v1y*v3f))) + (v3x*((v1y*v2f) - (v2y*v1f)))); } /*}}} */ /*{{{ void planarizip ( float *eqn,*/ void planarizip ( float *eqn, float *v1, float *v2, float *v3, int index, float fx1, float fx2, float fx3, float fy1, float fy2, float fy3, float fx23, float fx31, float fx12, float fC, int times ) { int i; register int opcode=0x501; *((int *) eqn++)=opcode; for (i=times; i; i-- ) { register float fv1=*v1++; register float fv2=*v2++; register float fv3=*v3++; *eqn++= fC * ((fy1 * (fv3 - fv2)) + (fy2 * (fv1 - fv3)) + (fy3 * (fv2 - fv1))); *eqn++=-fC*((fv1 * fx23) + (fv2 * fx31) + (fv3 * fx12)); *eqn++= fC*((fx1*((fy2*fv3) - (fy3*fv2))) + (fx2*((fy3*fv1) - (fy1*fv3))) + (fx3*((fy1*fv2) - (fy2*fv1)))); } } /*}}} */ /*{{{ variables for binning*/ binchunk *free_binchunks=NULL; screenbin *screen0bins=NULL, *screen1bins=NULL, *screenbins =NULL; int DMAscreen=0, writeScreen=1; /*}}} */ /*{{{ void grab_binchunks ( int grab_chunks )*/ void grab_binchunks ( int grab_chunks ) { /* mallocs and inits an initial tranche of binchunks */ int i; /* printf ("Grab_binchunks, needs %d\n", grab_chunks ); */ for (i=0; iusage=0; bin->next=free_binchunks; free_binchunks=bin; } } /*}}} */ /*{{{ binchunk *next_binchunk ()*/ binchunk *next_binchunk () { binchunk *chunky; if (free_binchunks == NULL) { grab_binchunks(8); } chunky=free_binchunks; chunky->usage=0; free_binchunks=free_binchunks->next; return chunky; } /*}}} */ /*{{{ void create_screenbins ( int screenx, int screeny )*/ void create_screenbins ( int screenx, int screeny ) { int binsx=screenx >> divpl5_xshift; int binsy=screeny >> divpl5_yshift; int i, j; printf ("create_screenbins, %d by %d\n", screenx, screeny ); screen0bins=(screenbin *) malloc (binsx*binsy*sizeof(screenbin)); screen1bins=(screenbin *) malloc (binsx*binsy*sizeof(screenbin)); printf ("grabbing screen0bins\n" ); for (i=0; ihead; if (chunk == NULL) { printf ("Error, initially empty region in liberate_screenbins\n" ); } chunk->usage=0; chunk=chunk->next; while (chunk) { binchunk *nxt=chunk->next; chunk->next=free_binchunks; chunk=nxt; } screenbins++; } } } /*}}} */ /*{{{ void binitize ( int macro_lo, int macro_hi,*/ void binitize ( int macro_lo, int macro_hi, float fminx, float fminy, float fmaxx, float fmaxy, int screen_maxx, int screen_maxy, int screen_bins_x ) { /* binitizes a primitive of known screen-space extents the DMA engine macros associated with the primitive are held in macro_lo, macro_hi - typically { SEND macro_address,size } the screen-space extents are held in fminx .. fmaxy, and the integer screen resolution is held in screen_maxx, screen_maxy, with (optimization) the bin-count in the x-direction held in screen_bins_x To binitize, we first work out what is the lower left corner bin, then outer loop in y, inner loop x, dropping the macro into all encountered bins. */ int minx, miny, maxx, maxy; minx=(int) fminx; miny=(int) fminy; maxx=(int) fmaxx; maxy=(int) fmaxy; if (maxx < 0) return; if (maxy < 0) return; if (minx > screen_maxx) return; if (miny > screen_maxy) return; minx >>= divpl5_xshift; miny >>= divpl5_yshift; maxx >>= divpl5_xshift; maxy >>= divpl5_yshift; /* so we have minimax xy in screen-space bin indices - put the data into bins */ { /* get 1st bin */ int screenbinix=(miny*screen_bins_x) + minx; screenbin *top_left_bin=&screenbins[screenbinix]; screenbin *lbin=top_left_bin; screenbin *xbin=lbin; register int x, y; /* scan down all y bins */ for (y=(maxy-miny)+1; y; y-- ) { /* scan across all x bins */ for (x=(maxx-minx)+1; x; x-- ) { /* add doubleword macro to bin */ register binchunk *bin=xbin->tail; register int usage=bin->usage; if (bin->usage == BIN_FULL) { binchunk *nextbin=next_binchunk (); bin->DMA_opcodes[usage++]=(int) nextbin; bin->DMA_opcodes[usage++]=DMA_GOTO; bin=nextbin; xbin->tail=bin; usage=0; } bin->DMA_opcodes[usage++]=macro_lo; bin->DMA_opcodes[usage++]=macro_hi; bin->usage=usage; xbin++; } lbin+=screen_bins_x; } } } /*}}} */ /*{{{ void safe_binitize ( int macro_lo, int macro_hi,*/ void safe_binitize ( int macro_lo, int macro_hi, float fminx, float fminy, float fmaxx, float fmaxy, int screen_bins_x ) { /* binitizes a primitive of known screen-space extents the DMA engine macros associated with the primitive are held in macro_lo, macro_hi - typically { SEND macro_address,size } the screen-space extents are held in fminx .. fmaxy, and the integer screen resolution is held in screen_maxx, screen_maxy, with (optimization) the bin-count in the x-direction held in screen_bins_x To binitize, we first work out what is the lower left corner bin, then outer loop in y, inner loop x, dropping the macro into all encountered bins. */ int minx, miny, maxx, maxy; minx=(int) fminx; miny=(int) fminy; maxx=(int) fmaxx; maxy=(int) fmaxy; minx >>= divpl5_xshift; miny >>= divpl5_yshift; maxx >>= divpl5_xshift; maxy >>= divpl5_yshift; /* so we have minimax xy in screen-space bin indices - put the data into bins */ { /* get 1st bin */ int screenbinix=(miny*screen_bins_x) + minx; screenbin *top_left_bin=&screenbins[screenbinix]; screenbin *lbin=top_left_bin; screenbin *xbin=lbin; register int x, y; /* scan down all y bins */ for (y=(maxy-miny)+1; y; y-- ) { /* scan across all x bins */ for (x=(maxx-minx)+1; x; x-- ) { /* add doubleword macro to bin */ register binchunk *bin=xbin->tail; register int usage=bin->usage; if (bin->usage == BIN_FULL) { binchunk *nextbin=next_binchunk (); bin->DMA_opcodes[usage++]=(int) nextbin; bin->DMA_opcodes[usage++]=DMA_GOTO; bin=nextbin; xbin->tail=bin; usage=0; } bin->DMA_opcodes[usage++]=macro_lo; bin->DMA_opcodes[usage++]=macro_hi; bin->usage=usage; xbin++; } lbin+=screen_bins_x; } } } /*}}} */