-/* $Id: extract.c,v 1.179 2005-04-28 08:20:39 adam Exp $
+/* $Id: extract.c,v 1.186 2005-06-14 20:28:54 adam Exp $
Copyright (C) 1995-2005
Index Data ApS
#include <ctype.h>
#ifdef WIN32
#include <io.h>
-#else
+#endif
+#if HAVE_UNISTD_H
#include <unistd.h>
#endif
#include <fcntl.h>
int attrSet, attrUse;
iscz1_decode(decode_handle, &dst, &src);
- assert(key.len < 4 && key.len > 2);
+ assert(key.len <= 4 && key.len > 2);
attrSet = (int) key.mem[0] >> 16;
attrUse = (int) key.mem[0] & 65535;
off_t file_moffset; /* offset of rec/rec boundary */
int file_more;
int fd;
- char *sdrbuf;
- int sdrmax;
};
static struct file_read_info *file_read_start (int fd)
fi->fd = fd;
fi->file_max = 0;
fi->file_moffset = 0;
- fi->sdrbuf = 0;
- fi->sdrmax = 0;
+ fi->file_offset = 0;
+ fi->file_more = 0;
return fi;
}
{
struct file_read_info *p = (struct file_read_info *) handle;
p->file_offset = offset;
- if (p->sdrbuf)
- return offset;
return lseek (p->fd, offset, SEEK_SET);
}
struct file_read_info *p = (struct file_read_info *) handle;
int fd = p->fd;
int r;
- if (p->sdrbuf)
- {
- r = count;
- if (r > p->sdrmax - p->file_offset)
- r = p->sdrmax - p->file_offset;
- if (r)
- memcpy (buf, p->sdrbuf + p->file_offset, r);
- }
- else
- r = read (fd, buf, count);
+ r = read (fd, buf, count);
if (r > 0)
{
p->file_offset += r;
return r;
}
-static void file_begin (void *handle)
-{
- struct file_read_info *p = (struct file_read_info *) handle;
-
- p->file_offset = p->file_moffset;
- if (!p->sdrbuf && p->file_moffset)
- lseek (p->fd, p->file_moffset, SEEK_SET);
- p->file_more = 0;
-}
-
static void file_end (void *handle, off_t offset)
{
struct file_read_info *p = (struct file_read_info *) handle;
- assert (p->file_more == 0);
- p->file_more = 1;
- p->file_moffset = offset;
+ if (offset != p->file_moffset)
+ {
+ p->file_moffset = offset;
+ p->file_more = 1;
+ }
}
static char *fileMatchStr (ZebraHandle zh,
keys->buf_used = 0;
iscz1_reset(keys->codec_handle);
}
-
+
+static void init_extractCtrl(ZebraHandle zh, struct recExtractCtrl *ctrl)
+{
+ int i;
+ for (i = 0; i<256; i++)
+ {
+ if (zebra_maps_is_positioned(zh->reg->zebra_maps, i))
+ ctrl->seqno[i] = 1;
+ else
+ ctrl->seqno[i] = 0;
+ }
+ ctrl->zebra_maps = zh->reg->zebra_maps;
+ ctrl->flagShowRecords = !zh->m_flag_rw;
+}
+
static int file_extract_record(ZebraHandle zh,
SYSNO *sysno, const char *fname,
int deleteFlag,
struct recExtractCtrl extractCtrl;
/* we are going to read from a file, so prepare the extraction */
- int i;
-
create_rec_keys_codec(&zh->reg->keys);
zh->reg->sortKeys.buf_used = 0;
-
recordOffset = fi->file_moffset;
extractCtrl.handle = zh;
extract_set_store_data_prepare(&extractCtrl);
- for (i = 0; i<256; i++)
- {
- if (zebra_maps_is_positioned(zh->reg->zebra_maps, i))
- extractCtrl.seqno[i] = 1;
- else
- extractCtrl.seqno[i] = 0;
- }
- extractCtrl.zebra_maps = zh->reg->zebra_maps;
- extractCtrl.flagShowRecords = !zh->m_flag_rw;
+ init_extractCtrl(zh, &extractCtrl);
if (!zh->m_flag_rw)
printf ("File: %s " PRINTF_OFF_T "\n", fname, recordOffset);
rec->size[recInfo_storeData] = zh->store_data_size;
rec->info[recInfo_storeData] = zh->store_data_buf;
zh->store_data_buf = 0;
- file_end(fi, fi->file_offset);
}
else if (zh->m_store_data)
{
fi = file_read_start (fd);
do
{
- file_begin (fi);
+ fi->file_moffset = fi->file_offset;
+ fi->file_more = 0; /* file_end not called (yet) */
r = file_extract_record (zh, sysno, fname, deleteFlag, fi, 1,
recType, recTypeClientData);
- } while (r && !sysno && fi->file_more);
+ if (fi->file_more)
+ { /* file_end has been called so reset offset .. */
+ fi->file_offset = fi->file_moffset;
+ lseek(fi->fd, fi->file_moffset, SEEK_SET);
+ }
+ }
+ while (r && !sysno);
file_read_stop (fi);
if (fd != -1)
close (fd);
{
RecordAttr *recordAttr;
struct recExtractCtrl extractCtrl;
- int i, r;
+ int r;
const char *matchStr = 0;
RecType recType = NULL;
void *clientData;
extractCtrl.schemaAdd = extract_schema_add;
extractCtrl.dh = zh->reg->dh;
extractCtrl.handle = zh;
- extractCtrl.zebra_maps = zh->reg->zebra_maps;
- extractCtrl.flagShowRecords = 0;
extractCtrl.match_criteria[0] = '\0';
- for (i = 0; i<256; i++)
- {
- if (zebra_maps_is_positioned(zh->reg->zebra_maps, i))
- extractCtrl.seqno[i] = 1;
- else
- extractCtrl.seqno[i] = 0;
- }
+
+ init_extractCtrl(zh, &extractCtrl);
+
extract_set_store_data_prepare(&extractCtrl);
r = (*recType->extract)(clientData, &extractCtrl);
/* update store data */
xfree (rec->info[recInfo_storeData]);
- if (zh->m_store_data)
+
+ /* update store data */
+ if (zh->store_data_buf)
+ {
+ rec->size[recInfo_storeData] = zh->store_data_size;
+ rec->info[recInfo_storeData] = zh->store_data_buf;
+ zh->store_data_buf = 0;
+ }
+ else if (zh->m_store_data)
{
rec->size[recInfo_storeData] = recordAttr->recordSize;
rec->info[recInfo_storeData] = (char *)
{
ZebraHandle zh = (ZebraHandle) handle;
struct recExtractCtrl extractCtrl;
- int i;
if (zebraExplain_curDatabase (zh->reg->zei,
rec->info[recInfo_databaseName]))
extractCtrl.tokenAdd = extract_token_add;
extractCtrl.schemaAdd = extract_schema_add;
extractCtrl.dh = zh->reg->dh;
- for (i = 0; i<256; i++)
- extractCtrl.seqno[i] = 0;
- extractCtrl.zebra_maps = zh->reg->zebra_maps;
+
+ init_extractCtrl(zh, &extractCtrl);
+
extractCtrl.flagShowRecords = 0;
extractCtrl.match_criteria[0] = '\0';
extractCtrl.handle = handle;
keys->buf_used = dst - keys->buf;
}
+ZEBRA_RES zebra_snippets_rec_keys(ZebraHandle zh, struct recKeys *reckeys,
+ zebra_snippets *snippets)
+{
+ void *decode_handle = iscz1_start();
+ int off = 0;
+ int seqno = 0;
+ NMEM nmem = nmem_create();
+
+ yaz_log(YLOG_LOG, "zebra_rec_keys_snippets buf=%p sz=%d", reckeys->buf,
+ reckeys->buf_used);
+ assert(reckeys->buf);
+ while (off < reckeys->buf_used)
+ {
+ const char *src = reckeys->buf + off;
+ struct it_key key;
+ char *dst = (char*) &key;
+ char dst_buf[IT_MAX_WORD];
+ char *dst_term = dst_buf;
+
+ iscz1_decode(decode_handle, &dst, &src);
+ assert(key.len <= 4 && key.len > 2);
+
+ seqno = (int) key.mem[key.len-1];
+
+ zebra_term_untrans_iconv(zh, nmem, src[0], &dst_term, src+1);
+ zebra_snippets_append(snippets, seqno, src[0], key.mem[0], dst_term);
+ while (*src++)
+ ;
+ off = src - reckeys->buf;
+ nmem_reset(nmem);
+ }
+ nmem_destroy(nmem);
+ iscz1_stop(decode_handle);
+ return ZEBRA_OK;
+}
+
+void print_rec_keys(ZebraHandle zh, struct recKeys *reckeys)
+{
+ void *decode_handle = iscz1_start();
+ int off = 0;
+ int seqno = 0;
+ NMEM nmem = nmem_create();
+
+ yaz_log(YLOG_LOG, "print_rec_keys buf=%p sz=%d", reckeys->buf,
+ reckeys->buf_used);
+ assert(reckeys->buf);
+ while (off < reckeys->buf_used)
+ {
+ const char *src = reckeys->buf + off;
+ struct it_key key;
+ char *dst = (char*) &key;
+ int attrSet, attrUse;
+ char dst_buf[IT_MAX_WORD];
+ char *dst_term = dst_buf;
+
+ iscz1_decode(decode_handle, &dst, &src);
+ assert(key.len <= 4 && key.len > 2);
+
+ attrSet = (int) key.mem[0] >> 16;
+ attrUse = (int) key.mem[0] & 65535;
+ seqno = (int) key.mem[key.len-1];
+
+ zebra_term_untrans_iconv(zh, nmem, src[0], &dst_term, src+1);
+
+ yaz_log(YLOG_LOG, "ord=" ZINT_FORMAT " seqno=%d term=%s",
+ key.mem[0], seqno, dst_term);
+ while (*src++)
+ ;
+ off = src - reckeys->buf;
+ nmem_reset(nmem);
+ }
+ nmem_destroy(nmem);
+ iscz1_stop(decode_handle);
+}
+
void extract_add_index_string (RecWord *p, const char *str, int length)
{
struct it_key key;