/* udbuild.c  - build udict & table */

#include "all.h"
#define UCHAR unsigned char

long lseek();

char *stat[228] = {
"A", /* 928 */
"C", /* 721 */
"D", /* 5494 */
"E", /* 5967 */
"G", /* 3836 */
"H", /* 733 */
"K", /* 526 */
"L", /* 1735 */
"M", /* 761 */
"N", /* 3115 */
"O", /* 371 */
"P", /* 332 */
"R", /* 2950 */
"S", /* 9498 */
"T", /* 2945 */
"W", /* 176 */
"Y", /* 4542 */
"AB", /* 781 */
"AC", /* 672 */
"AD", /* 417 */
"AG", /* 434 */
"AI", /* 436 */
"AK", /* 191 */
"AL", /* 2331 */
"AM", /* 392 */
"AN", /* 2103 */
"AP", /* 328 */
"AR", /* 1369 */
"AS", /* 621 */
"AT", /* 3590 */
"AY", /* 251 */
"BA", /* 233 */
"BE", /* 382 */
"BI", /* 380 */
"BL", /* 1028 */
"BO", /* 282 */
"BU", /* 150 */
"CA", /* 1028 */
"CE", /* 1313 */
"CH", /* 840 */
"CI", /* 825 */
"CK", /* 558 */
"CL", /* 184 */
"CO", /* 404 */
"CR", /* 206 */
"CT", /* 1037 */
"CU", /* 348 */
"CY", /* 164 */
"DA", /* 330 */
"DE", /* 1405 */
"DI", /* 901 */
"DL", /* 357 */
"DO", /* 253 */
"DS", /* 370 */
"DU", /* 156 */
"EA", /* 666 */
"EC", /* 627 */
"ED", /* 4878 */
"EE", /* 340 */
"EF", /* 160 */
"EG", /* 204 */
"EI", /* 147 */
"EL", /* 1177 */
"EM", /* 490 */
"EN", /* 3045 */
"EO", /* 163 */
"EP", /* 226 */
"ER", /* 5692 */
"ES", /* 4732 */
"ET", /* 1054 */
"EV", /* 154 */
"EY", /* 172 */
"FE", /* 273 */
"FF", /* 171 */
"FI", /* 449 */
"FO", /* 165 */
"FU", /* 325 */
"GA", /* 325 */
"GE", /* 1040 */
"GH", /* 327 */
"GI", /* 519 */
"GL", /* 340 */
"GN", /* 164 */
"GO", /* 168 */
"GR", /* 308 */
"GS", /* 275 */
"GU", /* 173 */
"HA", /* 366 */
"HE", /* 1101 */
"HI", /* 606 */
"HO", /* 451 */
"HT", /* 326 */
"IA", /* 1229 */
"IB", /* 340 */
"IC", /* 2201 */
"ID", /* 571 */
"IE", /* 1508 */
"IF", /* 421 */
"IG", /* 418 */
"IL", /* 940 */
"IM", /* 344 */
"IN", /* 5946 */
"IO", /* 2526 */
"IP", /* 346 */
"IR", /* 339 */
"IS", /* 1867 */
"IT", /* 1870 */
"IV", /* 767 */
"IZ", /* 575 */
"KE", /* 717 */
"KI", /* 295 */
"KL", /* 147 */
"KS", /* 241 */
"LA", /* 1143 */
"LD", /* 215 */
"LE", /* 3029 */
"LI", /* 2141 */
"LL", /* 1155 */
"LO", /* 760 */
"LS", /* 415 */
"LT", /* 200 */
"LU", /* 222 */
"LY", /* 2166 */
"MA", /* 813 */
"MB", /* 178 */
"ME", /* 1273 */
"MI", /* 757 */
"MM", /* 155 */
"MO", /* 306 */
"MP", /* 283 */
"MS", /* 215 */
"NA", /* 1111 */
"NC", /* 896 */
"ND", /* 952 */
"NE", /* 2169 */
"NG", /* 4413 */
"NI", /* 1224 */
"NK", /* 168 */
"NN", /* 163 */
"NO", /* 317 */
"NS", /* 1375 */
"NT", /* 2276 */
"OA", /* 150 */
"OC", /* 283 */
"OD", /* 274 */
"OG", /* 432 */
"OI", /* 207 */
"OL", /* 718 */
"OM", /* 582 */
"ON", /* 3487 */
"OO", /* 297 */
"OP", /* 362 */
"OR", /* 1770 */
"OS", /* 523 */
"OT", /* 380 */
"OU", /* 1053 */
"OV", /* 155 */
"OW", /* 275 */
"PA", /* 318 */
"PE", /* 796 */
"PH", /* 382 */
"PI", /* 444 */
"PL", /* 365 */
"PO", /* 423 */
"PP", /* 240 */
"PR", /* 200 */
"PS", /* 209 */
"PT", /* 302 */
"QU", /* 214 */
"RA", /* 1816 */
"RC", /* 252 */
"RD", /* 454 */
"RE", /* 2014 */
"RG", /* 174 */
"RI", /* 2100 */
"RL", /* 219 */
"RM", /* 341 */
"RN", /* 325 */
"RO", /* 932 */
"RR", /* 217 */
"RS", /* 1352 */
"RT", /* 664 */
"RU", /* 261 */
"RY", /* 574 */
"SA", /* 300 */
"SC", /* 270 */
"SE", /* 1376 */
"SH", /* 728 */
"SI", /* 1316 */
"SL", /* 282 */
"SM", /* 364 */
"SO", /* 390 */
"SP", /* 209 */
"SS", /* 1648 */
"ST", /* 2193 */
"SU", /* 191 */
"TA", /* 1295 */
"TE", /* 3770 */
"TH", /* 732 */
"TI", /* 4842 */
"TL", /* 466 */
"TO", /* 1009 */
"TR", /* 749 */
"TS", /* 995 */
"TT", /* 429 */
"TU", /* 547 */
"TY", /* 691 */
"UA", /* 299 */
"UC", /* 248 */
"UD", /* 186 */
"UE", /* 347 */
"UI", /* 259 */
"UL", /* 876 */
"UM", /* 425 */
"UN", /* 364 */
"UP", /* 163 */
"UR", /* 796 */
"US", /* 1273 */
"UT", /* 440 */
"VA", /* 274 */
"VE", /* 1283 */
"VI", /* 396 */
"WA", /* 216 */
"WE", /* 189 */
"YI", /* 159 */
"YS", /* 155 */
"ZA", /* 150 */
"ZE"}; /* 483 */

unsigned long 	ptable[4000];
UCHAR 			dbuf[8192],pdict[8192];
long 			loc,wtot,wdict;

UCHAR 			word[256],lastword[34],trans[34];
int 			first,second,oldfirst,oldsecond,third,oldthird;
unsigned short 	dindex;

int 			fudict,fudtable,fd = -1;

unsigned short 	ctable[28][28];

int				dictlen,dictindex = 0;

UCHAR 			pathdict[80];
unsigned short 	tsize = 0;

main(argc,argv)
int argc;
char **argv;
{
UCHAR *udict,*udtable,*p;
register int i,j,k,l;
int c1,ch,rtn,err;

	rtn = FALSE;
	udtable = (UCHAR *)"Udtable";
	udict = (UCHAR *)"Udict";
	
    pathdict[0] = 0x00;
    dindex = 0;
	puts("Uedit dictionary builder, (C) 1988 by Rick Stiles"); 
	puts("  usage: Udbuild <spelldirectory> <-ttable> <-ddict>");
	if (argc>1) {
		for (c1=1; c1<argc; c1++) {
			p = (UCHAR *)argv[c1];
			if (*p=='-') {
				p++;
				ch = *(p++);
				if (*p<'0') continue;
				if (ch=='t' || ch=='T') udtable = p;
				if (ch=='d' || ch=='D') udict = p;
			} else {
    			strcpy(pathdict,p);
    			i = strlen(pathdict);
    			if (i>1) {
        			j = pathdict[i-1];
        			if (j!=':' && j!='/')
						{ pathdict[i] = '/'; i++; pathdict[i] = 0x00; }
    			}
			}
		}
	}
    strcat(pathdict,"dict. ");
	unlink(udict);
	unlink(udtable);
    fudict = open(udict,O_CREAT | O_RDWR); 
	if (fudict== -1) {
			printf("Can't create %s\n",udict);
			goto BADOUT;
	}

    oldthird = third = oldfirst = oldsecond = first = second= -1;

    loc = 0L;
	wtot = 0L;

	for (i=0; i<28; i++) for (j=0; j<28; j++) ctable[i][j] = 0;
    for (i=0; i<228; i++) {
		p = (UCHAR *)stat[i];
		if (*p) *p -= ('A' - 2);
		if (*(p+1)) *(p+1) -= ('A' - 2);
	}

    for (i='A'; i<='Z'; i++) {
        if (!newdict(i)) goto BADOUT;
		wdict = 0L;
        lastword[0]=0x00;
        word[0]=0x00;
        while (getWord(i)) { wdict++; if (!buildTable()) goto BADOUT; }
		wtot += wdict;
    }
OUT:
	ptable[tsize] = (loc << 11L) | 0x0700;
    tsize++;

    if (fudict && dindex>0) if (write(fudict,dbuf,dindex)!=dindex) goto BADOUT;
    dindex = 0;
    
    fudtable = open(udtable,O_CREAT | O_RDWR);
    if (fudtable == -1) {
			printf("Can't create %s\n",udtable);
			goto BADOUT;
	}

ctable[0][0] = 65533;

if (write(fudtable,&(ctable[0][0]),sizeof(ctable))!=sizeof(ctable)) goto BADMSG;

    for (i=0; i<228; i++) if (write(fudtable,stat[i],2)!=2) goto BADMSG;

    if (write(fudtable,ptable,tsize << 2)!=(tsize<<2)) {
BADMSG:
    	close(fudtable);
		puts("No room for table");
		goto BADOUT;
	}
    close(fudtable);
    
    printf("UDICT: %ld, UDTABLE: %ld, WORDS: %ld\n",loc,(long)(tsize << 2L) + 456L
+ sizeof(ctable),wtot);
	rtn = TRUE;
		
BADOUT:
    if (fudict!=-1) close(fudict);
	if (fd!=-1) close(fd);
	if (!rtn) {
		unlink(udict);
		unlink(udtable);
	}
    exit(0);
}

getWord(cdict)
int cdict;
{
register UCHAR *p,*q;
register int   j;

MORE:
	p = pdict + dictindex;
    while ( dictindex < dictlen && *p==0x00 ) { dictindex++; p++; }
    if (dictindex>=dictlen) { 
		if (!readMore()) return(FALSE);
		goto MORE;
	}

MOREWORD:
    q = word;
    j = 0;
	p = pdict + dictindex;
    while (*p && j<255) {
        *(q++) = *(p++);
        j++;
		dictindex++;
		if ( dictindex >= dictlen ) {
			p = pdict;
			if (!readMore()) {
				if (j>0) break;
				return(FALSE);
			}
		}	
    }
	
    *q = 0x00;
    if (j>=32 && *p) {
		msg("long word: ",word);
        goto MOREWORD;
    }
	
    q = word;
    first = *(q++);
    second = *(q++);
    if (second) third = *q;
    else third = 0;
    if ((first + 'A' - 2)!=cdict) {
        if (cdict=='A' && first==1) goto TRYIT;
        goto BADORDER;
    }
TRYIT:
    if (first>oldfirst) goto RTN;
    if (first==oldfirst) {
        if (second>oldsecond) goto RTN;
        if (second==oldsecond && third>=oldthird) goto RTN;
    }
BADORDER:
    msg("Out of order and skipping:",word);
    goto MORE;
RTN:
    return(TRUE);
}

buildTable()
{
register UCHAR *p;
register int i,c1,c2;
unsigned long s;

    i = strlen(word);
	c1 = first;
	c2 = second;
	if (c1==0 || c2==0) goto NOSTORE;
    if (c1!=oldfirst || c2!=oldsecond || third!=oldthird) {
/*========
FLAGS:
0x40: len<3
0x20: len == 3
0x10: START or END of s-rgn
==================== */
		s = (loc << 11L) | third;
        if (i<3) s |= 0x400;
        else if (i==3) s |= 0x200;
		
		if (c1!=oldfirst || c2!=oldsecond) {
			ctable[c1][c2] = tsize + 1;
			s |= 0x100;
		}
		ptable[tsize] = s;
		tsize++;
    }
    if (i>3) { /* translate & store the word */
        i = transWord(i);
        if ((i + dindex)>=8192) {
            if (write(fudict,dbuf,dindex)!=dindex) {
				puts("No room for dictionary");
				return(FALSE);
			}
            dindex = 0;
        }
        movmem(trans,&(dbuf[dindex]),i);
        dindex += i;
        loc += i;
    }
NOSTORE:
    strcpy(lastword,word);
    oldfirst = first;
    oldsecond = second;
    oldthird = third;
	return(TRUE);
}

#define MINSTAT 0
#define MAXSTAT 227
#define MAXNULL 16
#define ADDSTAT 28
#define ZERO    0
#define APOSTROPHE 27
#define MINALPHA 1
#define MAXALPHA 26

transWord(len)
int len;
{
register UCHAR *p,*q;
register long bits;
register int   j,n,c1;
int k;
UCHAR hits[32];

    p = &(word[3]);
    len -= 3;       /* chopped off 1st 3 letters */
    bits = 0L;
    hits[len] = 0;
    for (j=0; j<len; j++) {
        k = pair(p[j],p[j+1]);
        hits[j] = k;
        if (k) bits |= (1L << j);
    }
    n = 0;
    q = trans;
    for (j=0; j<=len; j++) {
        c1 = p[j];
        if (bits & 1L) {                                     /* got a hit */
            if (bits & 2L) {
                if (bits & 4L) goto TAKETWO;
                goto TAKEONE;
            } else {
TAKETWO:
                q[n++] = hits[j++];
                bits >>= 1L;
            }
        } else {                                          /* must take it */
TAKEONE:
            q[n++] = c1;
            if (c1==0) break;
        }
        bits >>= 1L;
    }
    return(n);
}

pair(c1,c2)
register int c1,c2;
{
register UCHAR *r;
register int k;

    for (k=MINSTAT; k<=MAXSTAT; k++) {
        r = (UCHAR *)stat[k];
        if (*r==c1 && *(r+1)==c2)  return(k+ADDSTAT);
    }
    return(0);
}

newdict(cdict)
int cdict;
{
register int  i;

	if (fd!=-1) close(fd);
    i = strlen(pathdict);
    pathdict[i-1] = cdict;

    printf("%s\n",pathdict);
    fd = open(pathdict,O_RDONLY);
    if (fd == -1) {
		printf("Can't find %s\n",pathdict);
		return(FALSE);
	}
	readMore();
	return(TRUE);
}

readMore()
{
register UCHAR  *p;
register int  i,c,d;

	dictindex = 0;
	dictlen = 0;
    d = read(fd,pdict,8192);
	if (d<1) return(FALSE);
	dictlen = d;

    p = pdict;
	i = 0;
    while (1) {
        c = *p;
		if (c>='a' && c<='z') { c -= ('a' - 2); goto TAKEIT; }
		if (c>='A' && c<='Z') { c -= ('A' - 2); goto TAKEIT; }
        if (c<=' ') { c=0; goto TAKEIT; }
        if (c==39)  { c = 1; goto TAKEIT; }
        printf("Illegal char %c in %s\n",c, pathdict);
        c = 0;
TAKEIT:
        *p = c;
BUMPIT:
        if (++i>=dictlen) break;
        p++;
    }
    return(TRUE);
}

msg(x,y)
UCHAR *x,*y;
{
register UCHAR *z;
UCHAR zz[80];

	strcpy(zz,y);
	z = zz;
	while (*z) {
		if (*z==1) *z = 39;
		else *z += 63;
		z++;
	}
	printf("%s %s\n",x,zz);
}

