What is the mathematical basis of modulo fixed-point algorithm for IEEE 754 numbers

Question

At the end of the question, an code for calculating floating modulo operation (__ieee754_fmod ) from GNU C library is presented. I'm interested in the basic ideas behind this algorithm.

Of particular interest is the code marked / * fix point fmod * /.

    /* fix point fmod */
    n = ix - iy;
    while(n--) {
        hz=hx-hy;
        if(hz<0){hx = hx+hx;}
        else {
            if(hz==0)                /* return sign(x)*0 */
                return Zero[(uint64_t)sx>>63];
            hx = hz+hz;
        }
    }
    hz=hx-hy;
    if(hz>=0) {hx=hz;}

How does fixed-point arithmetic apply to floating-point numbers?
How it can be changed to apply to floating-point numbers in decimal notation?
Does it have some common with this algorithm for finding the remainder ?

while(x >= y){
    auto scaled_y = y;
    while(scaled_y*2 < x)
        scaled_y *= 2;
    x -= scaled_y;
}

Simplified examples or links to detailed descriptions are encouraged.

From GNU C library source code:

typedef union
{
    double value;
    struct
    {
        uint32_t lsw;
        uint32_t msw;
    } parts;
    uint64_t word;
} ieee_double_shape_type;


/* Get all in one, efficient on 64-bit machines.  */
#ifndef EXTRACT_WORDS64
# define EXTRACT_WORDS64(i,d)                                        \
do {                                                                \
  ieee_double_shape_type gh_u;                                        \
  gh_u.value = (d);                                                \
  (i) = gh_u.word;                                                \
} while (0)
#endif

/* Get all in one, efficient on 64-bit machines.  */
#ifndef INSERT_WORDS64
# define INSERT_WORDS64(d,i)                                        \
do {                                                                \
  ieee_double_shape_type iw_u;                                        \
  iw_u.word = (i);                                                \
  (d) = iw_u.value;                                                \
} while (0)
#endif


static const double one = 1.0, Zero[] = {0.0, -0.0,};
double __ieee754_fmod (double x, double y)
{
    int32_t n,ix,iy;
    int64_t hx,hy,hz,sx,i;
    EXTRACT_WORDS64(hx,x);
    EXTRACT_WORDS64(hy,y);
    sx = hx&UINT64_C(0x8000000000000000);        /* sign of x */
    hx ^=sx;                                /* |x| */
    hy &= UINT64_C(0x7fffffffffffffff);        /* |y| */
    /* purge off exception values */
    if(__builtin_expect(hy==0
                        || hx >= UINT64_C(0x7ff0000000000000)
                        || hy > UINT64_C(0x7ff0000000000000), 0))
        /* y=0,or x not finite or y is NaN */
        return (x*y)/(x*y);
    if(__builtin_expect(hx<=hy, 0)) {
        if(hx>63];        /* |x|=|y| return x*0*/
    }
    /* determine ix = ilogb(x) */
    if(__builtin_expect(hx0; i<<=1) ix -=1;
    } else ix = (hx>>52)-1023;
    /* determine iy = ilogb(y) */
    if(__builtin_expect(hy0; i<<=1) iy -=1;
    } else iy = (hy>>52)-1023;
    /* set up hx, hy and align y to x */
    if(__builtin_expect(ix >= -1022, 1))
        hx = UINT64_C(0x0010000000000000)|(UINT64_C(0x000fffffffffffff)&hx);
    else {                /* subnormal x, shift x to normal */
        n = -1022-ix;
        hx<<=n;
    }
    if(__builtin_expect(iy >= -1022, 1))
        hy = UINT64_C(0x0010000000000000)|(UINT64_C(0x000fffffffffffff)&hy);
    else {                /* subnormal y, shift y to normal */
        n = -1022-iy;
        hy<<=n;
    }
    /* fix point fmod */
    n = ix - iy;
    while(n--) {
        hz=hx-hy;
        if(hz<0){hx = hx+hx;}
        else {
            if(hz==0)                /* return sign(x)*0 */
                return Zero[(uint64_t)sx>>63];
            hx = hz+hz;
        }
    }
    hz=hx-hy;
    if(hz>=0) {hx=hz;}
    /* convert back to floating value and restore the sign */
    if(hx==0)                        /* return sign(x)*0 */
        return Zero[(uint64_t)sx>>63];
    while(hx= -1022, 1)) {        /* normalize output */
        hx = ((hx-UINT64_C(0x0010000000000000))|((uint64_t)(iy+1023)<<52));
        INSERT_WORDS64(x,hx|sx);
    } else {                /* subnormal output */
        n = -1022 - iy;
        hx>>=n;
        INSERT_WORDS64(x,hx|sx);
        x *= one;                /* create necessary signal */
    }
    return x;                /* exact output */
}

What is the mathematical basis of modulo fixed-point algorithm for IEEE 754 numbers

Answers (1)

Related Questions