* Sebastian Hammer, Adam Dickmeiss
*
* $Log: extract.c,v $
- * Revision 1.95 1999-05-21 12:00:17 adam
+ * Revision 1.100 2000-03-20 19:08:36 adam
+ * Added remote record import using Z39.50 extended services and Segment
+ * Requests.
+ *
+ * Revision 1.99 2000/02/24 10:57:02 adam
+ * Sequence number incremented after each incomplete-field.
+ *
+ * Revision 1.98 1999/09/07 07:19:21 adam
+ * Work on character mapping. Implemented replace rules.
+ *
+ * Revision 1.97 1999/07/06 12:28:04 adam
+ * Updated record index structure. Format includes version ID. Compression
+ * algorithm ID is stored for each record block.
+ *
+ * Revision 1.96 1999/05/26 07:49:13 adam
+ * C++ compilation.
+ *
+ * Revision 1.95 1999/05/21 12:00:17 adam
* Better diagnostics for extraction process.
*
* Revision 1.94 1999/05/20 12:57:18 adam
BFiles bfs = rGroup->bfs;
int rw = rGroup->flagRw;
data1_handle dh = rGroup->dh;
+ char *recordCompression;
+ int record_compression = REC_COMPRESS_NONE;
if (!mem)
mem = atoi(res_get_def (common_resource, "memMax", "4"))*1024*1024;
if (mem < 50000)
mem = 50000;
- key_buf = xmalloc (mem);
+ key_buf = (char **) xmalloc (mem);
ptr_top = mem/sizeof(char*);
ptr_i = 0;
return -1;
}
assert (!records);
- records = rec_open (bfs, rw);
+ recordCompression = res_get_def (common_resource,
+ "recordCompression", "none");
+ if (!strcmp (recordCompression, "none"))
+ record_compression = REC_COMPRESS_NONE;
+ if (!strcmp (recordCompression, "bzip2"))
+ record_compression = REC_COMPRESS_BZIP2;
+ records = rec_open (bfs, rw, record_compression);
if (!records)
{
dict_close (matchDict);
logf (LOG_LOG, "sorting section %d", key_file_no);
#if !SORT_EXTRA
qsort (key_buf + ptr_top-ptr_i, ptr_i, sizeof(char*), key_qsort_compare);
- getFnameTmp (out_fname, key_file_no);
+ getFnameTmp (common_resource, out_fname, key_file_no);
if (!(outf = fopen (out_fname, "wb")))
{
int rw = rGroup->flagRw;
if (rw)
zebraExplain_runNumberIncrement (zti, 1);
- zebraExplain_close (zti, rw, 0);
+ zebraExplain_close (zti, rw);
key_flush ();
xfree (key_buf);
rec_close (&records);
w->attrSet = VAL_BIB1;
w->attrUse = 1016;
w->reg_type = 'w';
+ w->extractCtrl = p;
}
static struct sortKey {
{
char *b;
- b = xmalloc (reckeys.buf_max += 128000);
+ b = (char *) xmalloc (reckeys.buf_max += 128000);
if (reckeys.buf_used > 0)
memcpy (b, reckeys.buf, reckeys.buf_used);
xfree (reckeys.buf);
if (sk->attrSet == p->attrSet && sk->attrUse == p->attrUse)
return;
- sk = xmalloc (sizeof(*sk));
+ sk = (struct sortKey *) xmalloc (sizeof(*sk));
sk->next = sortKeys;
sortKeys = sk;
- sk->string = xmalloc (length);
+ sk->string = (char *) xmalloc (length);
sk->length = length;
memcpy (sk->string, string, length);
return;
addString (p, buf, i);
}
+ (p->seqnos[p->reg_type])++; /* to separate this from next one */
}
static void addCompleteField (RecWord *p)
static void addRecordKey (RecWord *p)
{
+ WRBUF wrbuf;
+ if ((wrbuf = zebra_replace(p->zebra_maps, p->reg_type, 0,
+ p->string, p->length)))
+ {
+ p->string = wrbuf_buf(wrbuf);
+ p->length = wrbuf_len(wrbuf);
+ }
if (zebra_maps_is_complete (p->zebra_maps, p->reg_type))
addCompleteField (p);
else
static struct file_read_info *file_read_start (int fd)
{
- struct file_read_info *fi = xmalloc (sizeof(*fi));
+ struct file_read_info *fi = (struct file_read_info *)
+ xmalloc (sizeof(*fi));
fi->fd = fd;
fi->file_max = 0;
static off_t file_seek (void *handle, off_t offset)
{
- struct file_read_info *p = handle;
+ struct file_read_info *p = (struct file_read_info *) handle;
p->file_offset = offset;
if (p->sdrbuf)
return offset;
static off_t file_tell (void *handle)
{
- struct file_read_info *p = handle;
+ struct file_read_info *p = (struct file_read_info *) handle;
return p->file_offset;
}
static int file_read (void *handle, char *buf, size_t count)
{
- struct file_read_info *p = handle;
+ struct file_read_info *p = (struct file_read_info *) handle;
int fd = p->fd;
int r;
if (p->sdrbuf)
static void file_begin (void *handle)
{
- struct file_read_info *p = handle;
+ struct file_read_info *p = (struct file_read_info *) handle;
p->file_offset = p->file_moffset;
if (!p->sdrbuf && p->file_moffset)
static void file_end (void *handle, off_t offset)
{
- struct file_read_info *p = handle;
+ struct file_read_info *p = (struct file_read_info *) handle;
assert (p->file_more == 0);
p->file_more = 1;
static void recordLogPreamble (int level, const char *msg, void *info)
{
- struct recordLogInfo *p = info;
+ struct recordLogInfo *p = (struct recordLogInfo *) info;
FILE *outf = log_file ();
if (level & LOG_LOG)
extractCtrl.fh = fi;
extractCtrl.subType = subType;
extractCtrl.init = wordInit;
- extractCtrl.addWord = addRecordKey;
- extractCtrl.addSchema = addSchema;
+ extractCtrl.tokenAdd = addRecordKey;
+ extractCtrl.schemaAdd = addSchema;
extractCtrl.dh = rGroup->dh;
for (i = 0; i<256; i++)
{
if (rGroup->flagStoreData == 1)
{
rec->size[recInfo_storeData] = recordAttr->recordSize;
- rec->info[recInfo_storeData] = xmalloc (recordAttr->recordSize);
+ rec->info[recInfo_storeData] = (char *)
+ xmalloc (recordAttr->recordSize);
if (lseek (fi->fd, recordOffset, SEEK_SET) < 0)
{
logf (LOG_ERRNO|LOG_FATAL, "seek to %ld in %s",
{
if (zebraExplain_newDatabase (zti, rGroup->databaseName,
rGroup->explainDatabase))
- abort ();
+ return 0;
}
if (rGroup->flagStoreData == -1)
if (rGroup->flagStoreKeys == -1)
rGroup->flagStoreKeys = 0;
-#if ZEBRASDR
- if (rGroup->useSDR)
- {
- ZebraSdrHandle h;
- char xname[128], *xp;
-
- strncpy (xname, fname, 127);
- if (!(xp = strchr (xname, '.')))
- return 0;
- *xp = '\0';
- if (strcmp (xp+1, "sdr.bits"))
- return 0;
-
- h = zebraSdr_open (xname);
- if (!h)
- {
- logf (LOG_WARN, "sdr open %s", xname);
- return 0;
- }
- for (;;)
- {
- unsigned char *buf;
- char sdr_name[128];
- int r, segmentno;
-
- segmentno = zebraSdr_segment (h, 0);
- sprintf (sdr_name, "%%%s.%d", xname, segmentno);
-
-#if 0
- if (segmentno > 20)
- break;
-#endif
- r = zebraSdr_read (h, &buf);
-
- if (!r)
- break;
-
- fi = file_read_start (0);
- fi->sdrbuf = buf;
- fi->sdrmax = r;
- do
- {
- file_begin (fi);
- r = recordExtract (sysno, sdr_name, rGroup, deleteFlag, fi,
- recType, subType);
- } while (r && !sysno && fi->file_more);
- file_read_stop (fi);
- free (buf);
- }
- zebraSdr_close (h);
- return 1;
- }
-#endif
if (sysno && deleteFlag)
fd = -1;
else
reckeys.prevSeqNo = 0;
extractCtrl.init = wordInit;
- extractCtrl.addWord = addRecordKey;
- extractCtrl.addSchema = addSchema;
+ extractCtrl.tokenAdd = addRecordKey;
+ extractCtrl.schemaAdd = addSchema;
extractCtrl.dh = rGroup->dh;
for (i = 0; i<256; i++)
extractCtrl.seqno[i] = 0;