X-Git-Url: https://www.chiark.greenend.org.uk/ucgi/~yarrgweb/git?a=blobdiff_plain;f=pctb%2Focr.c;h=fafdba8eec2ac74f54c4ffc05f096f62c0c5d479;hb=8360f12a6a457b73ebb18dbeedbb15e6ed91318b;hp=5fdc4575ec4f64540129d00055e3847478f36a4f;hpb=89dfaeec1540f73ba85dbd25dd5332416f98778e;p=ypp-sc-tools.db-test.git diff --git a/pctb/ocr.c b/pctb/ocr.c index 5fdc457..fafdba8 100644 --- a/pctb/ocr.c +++ b/pctb/ocr.c @@ -1,5 +1,29 @@ /* - */ + * Core OCR algorithm (first exact bitmap match) + */ +/* + * This is part of ypp-sc-tools, a set of third-party tools for assisting + * players of Yohoho Puzzle Pirates. + * + * Copyright (C) 2009 Ian Jackson + * + * This program is free software: you can redistribute it and/or modify + * it under the terms of the GNU General Public License as published by + * the Free Software Foundation, either version 3 of the License, or + * (at your option) any later version. + * + * This program is distributed in the hope that it will be useful, + * but WITHOUT ANY WARRANTY; without even the implied warranty of + * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the + * GNU General Public License for more details. + * + * You should have received a copy of the GNU General Public License + * along with this program. If not, see . + * + * Yohoho and Puzzle Pirates are probably trademarks of Three Rings and + * are used without permission. This program is not endorsed or + * sponsored by Three Rings. + */ #include "ocr.h" @@ -22,10 +46,27 @@ static const char *context_names[]= { "Upper", "Digit" }; +struct OcrCellTypeInfo { + /* bitmaps of indices into context_names: */ + unsigned initial, nextword, midword; + int space_spaces; + const char *name; +}; +const struct OcrCellTypeInfo ocr_celltype_number= { + 4,4,4, + .space_spaces= 5, + .name= "number" +}; +const struct OcrCellTypeInfo ocr_celltype_text= { + .initial=2, /* Uppercase */ + .nextword=3, /* Either */ + .midword=1, /* Lower only */ + .space_spaces= 4, + .name= "text" +}; -#define NCONTEXTS (sizeof(context_names)/sizeof(context_names[0])) -#define SPACE_SPACES 4 +#define NCONTEXTS (sizeof(context_names)/sizeof(context_names[0])) struct OcrReader { int h; @@ -40,22 +81,7 @@ static int resolver_done; DEBUG_DEFINE_DEBUGF(ocr) -#define dbassert(x) \ - ((x) ? (void)0 : \ - fatal("Error in character set database.\n" \ - " Requirement not met: %s:%d: %s", __FILE__,__LINE__, #x)) - -static void fgetsline(FILE *f, char *lbuf, size_t lbufsz) { - errno=0; - char *s= fgets(lbuf,lbufsz,f); - sysassert(!ferror(f)); - dbassert(!feof(f)); - assert(s); - int l= strlen(lbuf); - dbassert(l>0); dbassert(lbuf[--l]='\n'); - lbuf[l]= 0; -} -#define FGETSLINE(f,buf) (fgetsline(f,buf,sizeof(buf))) +#define FGETSLINE (dbfile_getsline(lbuf,sizeof(lbuf),__FILE__,__LINE__)) static void cleardb_node(DatabaseNode *n) { int i; @@ -69,10 +95,9 @@ static void readdb(OcrReader *rd) { DatabaseNode *current, *additional; char chrs[MAXGLYPHCHRS+1]; Pixcol cv; - int r,j,ctxi; + int j,ctxi; int h, endsword; char lbuf[100]; - FILE *db; for (ctxi=0; ctxicontexts[ctxi]); @@ -81,41 +106,36 @@ static void readdb(OcrReader *rd) { asprintf(&dbfname,"%s/charset-%d.txt",get_vardir(),rd->h); sysassert(dbfname); - db= fopen(dbfname,"r"); - free(dbfname); - if (!db) { - sysassert(errno==ENOENT); - return; - } + if (!dbfile_open(dbfname)) + goto x; - FGETSLINE(db,lbuf); + FGETSLINE; dbassert(!strcmp(lbuf,"# ypp-sc-tools pctb font v1")); - r= fscanf(db, "%d", &h); - dbassert(r==1); + dbassert( dbfile_scanf("%d", &h) == 1); dbassert(h==rd->h); for (;;) { - FGETSLINE(db,lbuf); - if (!lbuf || lbuf[0]=='#') continue; + FGETSLINE; + if (!lbuf[0] || lbuf[0]=='#') continue; if (!strcmp(lbuf,".")) break; for (ctxi=0; ctxi0 && cr<=255); c= cr; } @@ -130,7 +150,7 @@ static void readdb(OcrReader *rd) { current= &rd->contexts[ctxi]; for (;;) { - FGETSLINE(db,lbuf); + FGETSLINE; if (!lbuf[0]) { dbassert(current != &rd->contexts[ctxi]); break; } char *ep; cv= strtoul(lbuf,&ep,16); dbassert(!*ep); @@ -164,8 +184,9 @@ static void readdb(OcrReader *rd) { strcpy(current->s, chrs); current->endsword= endsword; } - sysassert(!ferror(db)); - sysassert(!fclose(db)); + x: + dbfile_close(); + free(dbfname); } static void cu_pr_ctxmap(unsigned ctxmap) { @@ -187,6 +208,11 @@ static void callout_unknown(OcrReader *rd, int w, Pixcol cols[], const char *p; char cb; Pixcol pv; + + if (!o_resolver) + fatal("OCR failed - unrecognised characters or ligatures.\n" + "Character set database needs to be updated or augmented.\n" + "See README.charset.\n"); if (!resolver) { sysassert(! pipe(jobpipe) ); @@ -200,11 +226,11 @@ static void callout_unknown(OcrReader *rd, int w, Pixcol cols[], /* we know donepipe[1] is >= 4 and we have dealt with all the others * so we aren't in any danger of overwriting some other fd 4: */ sysassert( dup2(donepipe[1],4) ==4 ); - execlp("./yppsc-ocr-resolver", "yppsc-ocr-resolver", + execlp(o_resolver, o_resolver, DEBUGP(callout) ? "--debug" : "--noop-arg", "--automatic-1", (char*)0); - sysassert(!"execlp failed"); + sysassert(!"execlp ocr-resolver failed"); } sysassert(! close(jobpipe[0]) ); sysassert(! close(donepipe[1]) ); @@ -254,27 +280,10 @@ static void callout_unknown(OcrReader *rd, int w, Pixcol cols[], } if (r==0) { - pid_t pid; - int st; - for (;;) { - pid= waitpid(resolver_pid, &st, 0); - if (pid==-1) { sysassert(errno==EINTR); continue; } - break; - } - sysassert(pid==resolver_pid); - if (WIFEXITED(st)) { - if (WEXITSTATUS(st)) - fatal("character resolver failed with nonzero exit status %d", - WEXITSTATUS(st)); - fclose(resolver); - close(resolver_done); - resolver= 0; - } else if (WIFSIGNALED(st)) { - fatal("character resolver died due to signal %s%s", - strsignal(WTERMSIG(st)), WCOREDUMP(st)?" (core dumped)":""); - } else { - fatal("character resolver gave strange wait status %d",st); - } + waitpid_check_exitstatus(resolver_pid, "character resolver"); + fclose(resolver); + close(resolver_done); + resolver= 0; } else { assert(r==1); sysassert(cb==0); @@ -296,20 +305,6 @@ static void add_result(OcrReader *rd, const char *s, int l, int r, rd->nresults++; } -struct OcrCellTypeInfo { - unsigned initial, nextword, midword; - const char *name; -}; -const struct OcrCellTypeInfo ocr_celltype_number= { - 4,4,4, - .name= "number" -}; -const struct OcrCellTypeInfo ocr_celltype_text= { - .initial=2, /* Uppercase */ - .nextword=3, /* Either */ - .midword=1, /* Lower only */ - .name= "text" -}; const char *ocr_celltype_name(OcrCellType ct) { return ct->name; } @@ -338,7 +333,7 @@ OcrResultGlyph *ocr(OcrReader *rd, OcrCellType ct, int w, Pixcol cols[]) { if (!cols[x]) { nspaces++; x++; - if (nspaces==SPACE_SPACES) { + if (nspaces == ct->space_spaces) { debugf("OCR x=%x nspaces=%d space\n",x,nspaces); ctxmap= ct->nextword; } @@ -346,7 +341,7 @@ OcrResultGlyph *ocr(OcrReader *rd, OcrCellType ct, int w, Pixcol cols[]) { } /* something here, so we need to add the spaces */ - if (nspaces>=SPACE_SPACES) + if (nspaces >= ct->space_spaces) add_result(rd," ",x-nspaces,x+1,0); nspaces=0; @@ -411,7 +406,7 @@ OcrResultGlyph *ocr(OcrReader *rd, OcrCellType ct, int w, Pixcol cols[]) { if (uniquematch->s[0]) ctxmap= ct->midword; else debugf(" (empty)"); if (uniquematch->endsword) { - nspaces= SPACE_SPACES; + nspaces= ct->space_spaces; debugf("_"); ctxmap= ct->nextword; }