+#include "index.h"
+#include "orddict.h"
+#include <direntz.h>
+#include <charmap.h>
+
+#if _FILE_OFFSET_BITS == 64
+#define PRINTF_OFF_T "%Ld"
+#else
+#define PRINTF_OFF_T "%ld"
+#endif
+
+#define USE_SHELLSORT 0
+
+#if USE_SHELLSORT
+static void shellsort(void *ar, int r, size_t s,
+ int (*cmp)(const void *a, const void *b))
+{
+ char *a = ar;
+ char v[100];
+ int h, i, j, k;
+ static const int incs[16] = { 1391376, 463792, 198768, 86961, 33936,
+ 13776, 4592, 1968, 861, 336,
+ 112, 48, 21, 7, 3, 1 };
+ for ( k = 0; k < 16; k++)
+ for (h = incs[k], i = h; i < r; i++)
+ {
+ memcpy (v, a+s*i, s);
+ j = i;
+ while (j > h && (*cmp)(a + s*(j-h), v) > 0)
+ {
+ memcpy (a + s*j, a + s*(j-h), s);
+ j -= h;
+ }
+ memcpy (a+s*j, v, s);
+ }
+}
+#endif
+
+static void logRecord (ZebraHandle zh)
+{
+ ++zh->records_processed;
+ if (!(zh->records_processed % 1000))
+ {
+ yaz_log (YLOG_LOG, "Records: "ZINT_FORMAT" i/u/d "
+ ZINT_FORMAT"/"ZINT_FORMAT"/"ZINT_FORMAT,
+ zh->records_processed, zh->records_inserted, zh->records_updated,
+ zh->records_deleted);
+ }
+}
+
+static void extract_set_store_data_prepare(struct recExtractCtrl *p);
+
+static void extract_init (struct recExtractCtrl *p, RecWord *w)
+{
+ w->zebra_maps = p->zebra_maps;
+ w->seqno = 1;
+#if NATTR
+#else
+ w->attrSet = VAL_BIB1;
+ w->attrUse = 1016;
+#endif
+ w->index_name = 0;
+ w->index_type = 'w';
+ w->extractCtrl = p;
+ w->record_id = 0;
+ w->section_id = 0;
+}
+
+static void searchRecordKey(ZebraHandle zh,
+ zebra_rec_keys_t reckeys,
+ int attrSetS, int attrUseS,
+ const char **ws, int ws_length)
+{
+ int i;
+ int ch;
+
+ for (i = 0; i<ws_length; i++)
+ ws[i] = NULL;
+
+ ch = zebraExplain_lookup_attr_su_any_index(zh->reg->zei,
+ attrSetS, attrUseS);
+ if (ch < 0)
+ return ;
+
+ if (zebra_rec_keys_rewind(reckeys))
+ {
+ int startSeq = -1;
+ const char *str;
+ size_t slen;
+ struct it_key key;
+ zint seqno;
+ while (zebra_rec_keys_read(reckeys, &str, &slen, &key))
+ {
+ assert(key.len <= 4 && key.len > 2);
+
+ seqno = key.mem[key.len-1];
+
+ if (key.mem[0] == ch)
+ {
+ int woff;
+
+ if (startSeq == -1)
+ startSeq = seqno;
+ woff = seqno - startSeq;
+ if (woff >= 0 && woff < ws_length)
+ ws[woff] = str;
+ }
+ }
+ }
+}
+
+struct file_read_info {
+ off_t file_max; /* maximum offset so far */
+ off_t file_offset; /* current offset */
+ off_t file_moffset; /* offset of rec/rec boundary */
+ int file_more;
+ int fd;
+};
+
+static struct file_read_info *file_read_start (int fd)
+{
+ struct file_read_info *fi = (struct file_read_info *)
+ xmalloc (sizeof(*fi));
+
+ fi->fd = fd;
+ fi->file_max = 0;
+ fi->file_moffset = 0;
+ fi->file_offset = 0;
+ fi->file_more = 0;
+ return fi;
+}
+
+static void file_read_stop (struct file_read_info *fi)
+{
+ xfree (fi);
+}
+
+static off_t file_seek (void *handle, off_t offset)
+{
+ struct file_read_info *p = (struct file_read_info *) handle;
+ p->file_offset = offset;
+ return lseek (p->fd, offset, SEEK_SET);
+}
+
+static off_t file_tell (void *handle)
+{
+ struct file_read_info *p = (struct file_read_info *) handle;
+ return p->file_offset;
+}
+
+static int file_read (void *handle, char *buf, size_t count)
+{
+ struct file_read_info *p = (struct file_read_info *) handle;
+ int fd = p->fd;
+ int r;
+ r = read (fd, buf, count);
+ if (r > 0)
+ {
+ p->file_offset += r;
+ if (p->file_offset > p->file_max)
+ p->file_max = p->file_offset;
+ }
+ return r;
+}
+
+static void file_end (void *handle, off_t offset)
+{
+ struct file_read_info *p = (struct file_read_info *) handle;
+
+ if (offset != p->file_moffset)
+ {
+ p->file_moffset = offset;
+ p->file_more = 1;
+ }
+}
+
+#define FILE_MATCH_BLANK "\t "
+
+static char *fileMatchStr (ZebraHandle zh,
+ zebra_rec_keys_t reckeys,
+ const char *fname, const char *spec)
+{
+ static char dstBuf[2048]; /* static here ??? */
+ char *dst = dstBuf;
+ const char *s = spec;
+
+ while (1)
+ {
+ for (; *s && strchr(FILE_MATCH_BLANK, *s); s++)
+ ;
+ if (!*s)
+ break;
+ if (*s == '(')
+ {
+ const char *ws[32];
+ char attset_str[64], attname_str[64];
+ data1_attset *attset;
+ int i;
+ int attSet = 1, attUse = 1;
+ int first = 1;
+
+ for (s++; strchr(FILE_MATCH_BLANK, *s); s++)
+ ;
+ for (i = 0; *s && *s != ',' && *s != ')' &&
+ !strchr(FILE_MATCH_BLANK, *s); s++)
+ if (i+1 < sizeof(attset_str))
+ attset_str[i++] = *s;
+ attset_str[i] = '\0';
+
+ for (; strchr(FILE_MATCH_BLANK, *s); s++)
+ ;
+ if (*s == ',')
+ {
+ for (s++; strchr(FILE_MATCH_BLANK, *s); s++)
+ ;
+ for (i = 0; *s && *s != ')' &&
+ !strchr(FILE_MATCH_BLANK, *s); s++)
+ if (i+1 < sizeof(attname_str))
+ attname_str[i++] = *s;
+ attname_str[i] = '\0';
+ }
+
+ if ((attset = data1_get_attset (zh->reg->dh, attset_str)))
+ {
+ data1_att *att;
+ attSet = attset->reference;
+ att = data1_getattbyname(zh->reg->dh, attset, attname_str);
+ if (att)
+ attUse = att->value;
+ else
+ attUse = atoi (attname_str);
+ }
+ searchRecordKey (zh, reckeys, attSet, attUse, ws, 32);
+
+ if (*s != ')')
+ {
+ yaz_log (YLOG_WARN, "Missing ) in match criteria %s in group %s",
+ spec, zh->m_group ? zh->m_group : "none");
+ return NULL;
+ }
+ s++;
+
+ for (i = 0; i<32; i++)
+ if (ws[i])
+ {
+ if (first)
+ {
+ *dst++ = ' ';
+ first = 0;
+ }
+ strcpy (dst, ws[i]);
+ dst += strlen(ws[i]);
+ }
+ if (first)
+ {
+ yaz_log (YLOG_WARN, "Record didn't contain match"
+ " fields in (%s,%s)", attset_str, attname_str);
+ return NULL;
+ }
+ }
+ else if (*s == '$')
+ {
+ int spec_len;
+ char special[64];
+ const char *spec_src = NULL;
+ const char *s1 = ++s;
+ while (*s1 && !strchr(FILE_MATCH_BLANK, *s1))
+ s1++;
+
+ spec_len = s1 - s;
+ if (spec_len > sizeof(special)-1)
+ spec_len = sizeof(special)-1;
+ memcpy (special, s, spec_len);
+ special[spec_len] = '\0';
+ s = s1;
+
+ if (!strcmp (special, "group"))
+ spec_src = zh->m_group;
+ else if (!strcmp (special, "database"))
+ spec_src = zh->basenames[0];
+ else if (!strcmp (special, "filename")) {
+ spec_src = fname;
+ }
+ else if (!strcmp (special, "type"))
+ spec_src = zh->m_record_type;
+ else
+ spec_src = NULL;
+ if (spec_src)
+ {
+ strcpy (dst, spec_src);
+ dst += strlen (spec_src);
+ }
+ }
+ else if (*s == '\"' || *s == '\'')
+ {
+ int stopMarker = *s++;
+ char tmpString[64];
+ int i = 0;
+
+ while (*s && *s != stopMarker)
+ {
+ if (i+1 < sizeof(tmpString))
+ tmpString[i++] = *s++;
+ }
+ if (*s)
+ s++;
+ tmpString[i] = '\0';
+ strcpy (dst, tmpString);
+ dst += strlen (tmpString);
+ }
+ else
+ {
+ yaz_log (YLOG_WARN, "Syntax error in match criteria %s in group %s",
+ spec, zh->m_group ? zh->m_group : "none");
+ return NULL;
+ }
+ *dst++ = 1;
+ }
+ if (dst == dstBuf)
+ {
+ yaz_log (YLOG_WARN, "No match criteria for record %s in group %s",
+ fname, zh->m_group ? zh->m_group : "none");
+ return NULL;
+ }
+ *dst = '\0';
+ return dstBuf;
+}
+
+struct recordLogInfo {
+ const char *fname;
+ int recordOffset;
+ struct recordGroup *rGroup;
+};
+
+static void init_extractCtrl(ZebraHandle zh, struct recExtractCtrl *ctrl)
+{
+ int i;
+ for (i = 0; i<256; i++)
+ {
+ if (zebra_maps_is_positioned(zh->reg->zebra_maps, i))
+ ctrl->seqno[i] = 1;
+ else
+ ctrl->seqno[i] = 0;
+ }
+ ctrl->zebra_maps = zh->reg->zebra_maps;
+ ctrl->flagShowRecords = !zh->m_flag_rw;
+}
+
+static int file_extract_record(ZebraHandle zh,
+ SYSNO *sysno, const char *fname,
+ int deleteFlag,
+ struct file_read_info *fi,
+ int force_update,
+ RecType recType,
+ void *recTypeClientData)
+{
+ RecordAttr *recordAttr;
+ int r;
+ const char *matchStr = 0;
+ SYSNO sysnotmp;
+ Record rec;
+ off_t recordOffset = 0;
+ struct recExtractCtrl extractCtrl;
+
+ /* announce database */
+ if (zebraExplain_curDatabase (zh->reg->zei, zh->basenames[0]))
+ {
+ if (zebraExplain_newDatabase (zh->reg->zei, zh->basenames[0],
+ zh->m_explain_database))
+ return 0;
+ }
+
+ if (fi->fd != -1)
+ {
+ /* we are going to read from a file, so prepare the extraction */
+ zebra_rec_keys_reset(zh->reg->keys);
+
+#if NATTR
+ zebra_rec_keys_reset(zh->reg->sortKeys);
+#else
+ zh->reg->sortKeys.buf_used = 0;
+#endif
+ recordOffset = fi->file_moffset;
+ extractCtrl.handle = zh;
+ extractCtrl.offset = fi->file_moffset;
+ extractCtrl.readf = file_read;
+ extractCtrl.seekf = file_seek;
+ extractCtrl.tellf = file_tell;
+ extractCtrl.endf = file_end;
+ extractCtrl.fh = fi;
+ extractCtrl.init = extract_init;
+ extractCtrl.tokenAdd = extract_token_add;
+ extractCtrl.schemaAdd = extract_schema_add;
+ extractCtrl.dh = zh->reg->dh;
+ extractCtrl.match_criteria[0] = '\0';
+ extractCtrl.staticrank = 0;
+
+ extractCtrl.first_record = fi->file_offset ? 0 : 1;
+
+ extract_set_store_data_prepare(&extractCtrl);
+
+ init_extractCtrl(zh, &extractCtrl);
+
+ if (!zh->m_flag_rw)
+ printf ("File: %s " PRINTF_OFF_T "\n", fname, recordOffset);
+ if (zh->m_flag_rw)
+ {
+ char msg[512];
+ sprintf (msg, "%s:" PRINTF_OFF_T , fname, recordOffset);
+ yaz_log_init_prefix2 (msg);
+ }
+
+ r = (*recType->extract)(recTypeClientData, &extractCtrl);
+
+ yaz_log_init_prefix2 (0);
+ if (r == RECCTRL_EXTRACT_EOF)
+ return 0;
+ else if (r == RECCTRL_EXTRACT_ERROR_GENERIC)
+ {
+ /* error occured during extraction ... */
+ if (zh->m_flag_rw &&
+ zh->records_processed < zh->m_file_verbose_limit)
+ {
+ yaz_log (YLOG_WARN, "fail %s %s " PRINTF_OFF_T, zh->m_record_type,
+ fname, recordOffset);
+ }
+ return 0;
+ }
+ else if (r == RECCTRL_EXTRACT_ERROR_NO_SUCH_FILTER)
+ {
+ /* error occured during extraction ... */
+ if (zh->m_flag_rw &&
+ zh->records_processed < zh->m_file_verbose_limit)
+ {
+ yaz_log (YLOG_WARN, "no filter for %s %s "
+ PRINTF_OFF_T, zh->m_record_type,
+ fname, recordOffset);
+ }
+ return 0;
+ }
+ if (extractCtrl.match_criteria[0])
+ matchStr = extractCtrl.match_criteria;
+ }
+
+ /* perform match if sysno not known and if match criteria is specified */
+ if (!sysno)
+ {
+ sysnotmp = 0;
+ sysno = &sysnotmp;
+
+ if (matchStr == 0 && zh->m_record_id && *zh->m_record_id)
+ {
+ matchStr = fileMatchStr (zh, zh->reg->keys, fname,
+ zh->m_record_id);
+ if (!matchStr)
+ {
+ yaz_log(YLOG_WARN, "Bad match criteria");
+ return 0;
+ }
+ }
+ if (matchStr)
+ {
+ int db_ord = zebraExplain_get_database_ord(zh->reg->zei);
+ char *rinfo = dict_lookup_ord(zh->reg->matchDict, db_ord,
+ matchStr);
+ if (rinfo)
+ {
+ assert(*rinfo == sizeof(*sysno));
+ memcpy (sysno, rinfo+1, sizeof(*sysno));
+ }
+ }
+ }
+ if (! *sysno && zebra_rec_keys_empty(zh->reg->keys) )
+ {
+ /* the extraction process returned no information - the record
+ is probably empty - unless flagShowRecords is in use */
+ if (!zh->m_flag_rw)
+ return 1;
+
+ if (zh->records_processed < zh->m_file_verbose_limit)
+ yaz_log (YLOG_WARN, "empty %s %s " PRINTF_OFF_T, zh->m_record_type,
+ fname, recordOffset);
+ return 1;
+ }
+
+ if (! *sysno)
+ {
+ /* new record */
+ if (deleteFlag)
+ {
+ yaz_log (YLOG_LOG, "delete %s %s " PRINTF_OFF_T, zh->m_record_type,
+ fname, recordOffset);
+ yaz_log (YLOG_WARN, "cannot delete record above (seems new)");
+ return 1;
+ }
+ if (zh->records_processed < zh->m_file_verbose_limit)
+ yaz_log (YLOG_LOG, "add %s %s " PRINTF_OFF_T, zh->m_record_type,
+ fname, recordOffset);
+ rec = rec_new (zh->reg->records);
+
+ *sysno = rec->sysno;
+
+ recordAttr = rec_init_attr (zh->reg->zei, rec);
+ recordAttr->staticrank = extractCtrl.staticrank;
+
+ if (matchStr)
+ {
+ int db_ord = zebraExplain_get_database_ord(zh->reg->zei);
+ dict_insert_ord(zh->reg->matchDict, db_ord, matchStr,
+ sizeof(*sysno), sysno);
+ }
+#if NATTR
+ extract_flushSortKeys (zh, *sysno, 1, zh->reg->sortKeys);
+#else
+ extract_flushSortKeys (zh, *sysno, 1, &zh->reg->sortKeys);
+#endif
+ extract_flushRecordKeys (zh, *sysno, 1, zh->reg->keys,
+ recordAttr->staticrank);
+ zh->records_inserted++;
+ }
+ else
+ {
+ /* record already exists */
+ zebra_rec_keys_t delkeys = zebra_rec_keys_open();
+
+#if NATTR
+ zebra_rec_keys_t sortKeys = zebra_rec_keys_open();
+#else
+ struct sortKeys sortKeys;
+#endif
+
+ rec = rec_get (zh->reg->records, *sysno);
+ assert (rec);
+
+ recordAttr = rec_init_attr (zh->reg->zei, rec);
+
+ zebra_rec_keys_set_buf(delkeys,
+ rec->info[recInfo_delKeys],
+ rec->size[recInfo_delKeys],
+ 0);
+
+#if NATTR
+ zebra_rec_keys_set_buf(sortKeys,
+ rec->info[recInfo_sortKeys],
+ rec->size[recInfo_sortKeys],
+ 0);
+ extract_flushSortKeys (zh, *sysno, 0, sortKeys);
+#else
+ sortKeys.buf_used = rec->size[recInfo_sortKeys];
+ sortKeys.buf = rec->info[recInfo_sortKeys];
+ extract_flushSortKeys (zh, *sysno, 0, &sortKeys);
+#endif
+
+ extract_flushRecordKeys (zh, *sysno, 0, delkeys,
+ recordAttr->staticrank); /* old values */
+ if (deleteFlag)
+ {
+ /* record going to be deleted */
+ if (zebra_rec_keys_empty(delkeys))
+ {
+ yaz_log (YLOG_LOG, "delete %s %s " PRINTF_OFF_T,
+ zh->m_record_type, fname, recordOffset);
+ yaz_log (YLOG_WARN, "cannot delete file above, storeKeys false (1)");
+ }
+ else
+ {
+ if (zh->records_processed < zh->m_file_verbose_limit)
+ yaz_log (YLOG_LOG, "delete %s %s " PRINTF_OFF_T,
+ zh->m_record_type, fname, recordOffset);
+ zh->records_deleted++;
+ if (matchStr)
+ {
+ int db_ord = zebraExplain_get_database_ord(zh->reg->zei);
+ dict_delete_ord(zh->reg->matchDict, db_ord, matchStr);
+ }
+ rec_del (zh->reg->records, &rec);
+ }
+ rec_rm (&rec);
+ logRecord (zh);
+ return 1;
+ }
+ else
+ {
+ /* flush new keys for sort&search etc */
+ if (zh->records_processed < zh->m_file_verbose_limit)
+ yaz_log (YLOG_LOG, "update %s %s " PRINTF_OFF_T,
+ zh->m_record_type, fname, recordOffset);
+ recordAttr->staticrank = extractCtrl.staticrank;
+#if NATTR
+ extract_flushSortKeys (zh, *sysno, 1, zh->reg->sortKeys);
+#else
+ extract_flushSortKeys (zh, *sysno, 1, &zh->reg->sortKeys);
+#endif
+ extract_flushRecordKeys (zh, *sysno, 1, zh->reg->keys,
+ recordAttr->staticrank);
+ zh->records_updated++;
+ }
+ zebra_rec_keys_close(delkeys);
+#if NATTR
+ zebra_rec_keys_close(sortKeys);
+#endif
+ }
+ /* update file type */
+ xfree (rec->info[recInfo_fileType]);
+ rec->info[recInfo_fileType] =
+ rec_strdup (zh->m_record_type, &rec->size[recInfo_fileType]);
+
+ /* update filename */
+ xfree (rec->info[recInfo_filename]);
+ rec->info[recInfo_filename] =
+ rec_strdup (fname, &rec->size[recInfo_filename]);
+
+ /* update delete keys */
+ xfree (rec->info[recInfo_delKeys]);
+ if (!zebra_rec_keys_empty(zh->reg->keys) && zh->m_store_keys == 1)
+ {
+ zebra_rec_keys_get_buf(zh->reg->keys,
+ &rec->info[recInfo_delKeys],
+ &rec->size[recInfo_delKeys]);
+ }
+ else
+ {
+ rec->info[recInfo_delKeys] = NULL;
+ rec->size[recInfo_delKeys] = 0;
+ }
+
+ /* update sort keys */
+ xfree (rec->info[recInfo_sortKeys]);
+
+#if NATTR
+ zebra_rec_keys_get_buf(zh->reg->sortKeys,
+ &rec->info[recInfo_sortKeys],
+ &rec->size[recInfo_sortKeys]);
+#else
+ rec->size[recInfo_sortKeys] = zh->reg->sortKeys.buf_used;
+ rec->info[recInfo_sortKeys] = zh->reg->sortKeys.buf;
+ zh->reg->sortKeys.buf = NULL;
+ zh->reg->sortKeys.buf_max = 0;
+#endif
+
+ /* save file size of original record */
+ zebraExplain_recordBytesIncrement (zh->reg->zei,
+ - recordAttr->recordSize);
+ recordAttr->recordSize = fi->file_moffset - recordOffset;
+ if (!recordAttr->recordSize)
+ recordAttr->recordSize = fi->file_max - recordOffset;
+ zebraExplain_recordBytesIncrement (zh->reg->zei,
+ recordAttr->recordSize);
+
+ /* set run-number for this record */
+ recordAttr->runNumber = zebraExplain_runNumberIncrement (zh->reg->zei,
+ 0);
+
+ /* update store data */
+ xfree (rec->info[recInfo_storeData]);
+ if (zh->store_data_buf)
+ {
+ rec->size[recInfo_storeData] = zh->store_data_size;
+ rec->info[recInfo_storeData] = zh->store_data_buf;
+ zh->store_data_buf = 0;
+ }
+ else if (zh->m_store_data)
+ {
+ rec->size[recInfo_storeData] = recordAttr->recordSize;
+ rec->info[recInfo_storeData] = (char *)
+ xmalloc (recordAttr->recordSize);
+ if (lseek (fi->fd, recordOffset, SEEK_SET) < 0)
+ {
+ yaz_log (YLOG_ERRNO|YLOG_FATAL, "seek to " PRINTF_OFF_T " in %s",
+ recordOffset, fname);
+ exit (1);
+ }
+ if (read (fi->fd, rec->info[recInfo_storeData], recordAttr->recordSize)
+ < recordAttr->recordSize)
+ {
+ yaz_log (YLOG_ERRNO|YLOG_FATAL, "read %d bytes of %s",
+ recordAttr->recordSize, fname);
+ exit (1);
+ }
+ }
+ else
+ {
+ rec->info[recInfo_storeData] = NULL;
+ rec->size[recInfo_storeData] = 0;
+ }
+ /* update database name */
+ xfree (rec->info[recInfo_databaseName]);
+ rec->info[recInfo_databaseName] =
+ rec_strdup (zh->basenames[0], &rec->size[recInfo_databaseName]);
+
+ /* update offset */
+ recordAttr->recordOffset = recordOffset;
+
+ /* commit this record */
+ rec_put (zh->reg->records, &rec);
+ logRecord (zh);
+ return 1;
+}
+
+int fileExtract (ZebraHandle zh, SYSNO *sysno, const char *fname,
+ int deleteFlag)
+{
+ int r, i, fd;
+ char gprefix[128];
+ char ext[128];
+ char ext_res[128];
+ struct file_read_info *fi;
+ const char *original_record_type = 0;
+ RecType recType;
+ void *recTypeClientData;
+
+ if (!zh->m_group || !*zh->m_group)
+ *gprefix = '\0';
+ else
+ sprintf (gprefix, "%s.", zh->m_group);
+
+ yaz_log (YLOG_DEBUG, "fileExtract %s", fname);
+
+ /* determine file extension */
+ *ext = '\0';
+ for (i = strlen(fname); --i >= 0; )
+ if (fname[i] == '/')
+ break;
+ else if (fname[i] == '.')
+ {
+ strcpy (ext, fname+i+1);
+ break;
+ }
+ /* determine file type - depending on extension */
+ original_record_type = zh->m_record_type;
+ if (!zh->m_record_type)
+ {
+ sprintf (ext_res, "%srecordType.%s", gprefix, ext);
+ zh->m_record_type = res_get (zh->res, ext_res);
+ }
+ if (!zh->m_record_type)
+ {
+ if (zh->records_processed < zh->m_file_verbose_limit)
+ yaz_log (YLOG_LOG, "? %s", fname);
+ return 0;
+ }
+ /* determine match criteria */
+ if (!zh->m_record_id)
+ {
+ sprintf (ext_res, "%srecordId.%s", gprefix, ext);
+ zh->m_record_id = res_get (zh->res, ext_res);
+ }
+
+ if (!(recType =
+ recType_byName (zh->reg->recTypes, zh->res, zh->m_record_type,
+ &recTypeClientData)))
+ {
+ yaz_log(YLOG_WARN, "No such record type: %s", zh->m_record_type);
+ return 0;
+ }
+
+ switch(recType->version)
+ {
+ case 0:
+ break;
+ default:
+ yaz_log(YLOG_WARN, "Bad filter version: %s", zh->m_record_type);
+ }
+ if (sysno && deleteFlag)
+ fd = -1;
+ else
+ {
+ char full_rep[1024];
+
+ if (zh->path_reg && !yaz_is_abspath (fname))
+ {
+ strcpy (full_rep, zh->path_reg);
+ strcat (full_rep, "/");
+ strcat (full_rep, fname);
+ }
+ else
+ strcpy (full_rep, fname);
+
+
+ if ((fd = open (full_rep, O_BINARY|O_RDONLY)) == -1)
+ {
+ yaz_log (YLOG_WARN|YLOG_ERRNO, "open %s", full_rep);
+ zh->m_record_type = original_record_type;
+ return 0;
+ }
+ }
+ fi = file_read_start (fd);
+ do
+ {
+ fi->file_moffset = fi->file_offset;
+ fi->file_more = 0; /* file_end not called (yet) */
+ r = file_extract_record (zh, sysno, fname, deleteFlag, fi, 1,
+ recType, recTypeClientData);
+ if (fi->file_more)
+ { /* file_end has been called so reset offset .. */
+ fi->file_offset = fi->file_moffset;
+ lseek(fi->fd, fi->file_moffset, SEEK_SET);
+ }
+ }
+ while (r && !sysno);
+ file_read_stop (fi);
+ if (fd != -1)
+ close (fd);
+ zh->m_record_type = original_record_type;
+ return r;
+}
+
+/*
+ If sysno is provided, then it's used to identify the reocord.
+ If not, and match_criteria is provided, then sysno is guessed
+ If not, and a record is provided, then sysno is got from there
+
+ */
+ZEBRA_RES buffer_extract_record(ZebraHandle zh,
+ const char *buf, size_t buf_size,
+ int delete_flag,
+ int test_mode,
+ const char *recordType,
+ SYSNO *sysno,
+ const char *match_criteria,
+ const char *fname,
+ int force_update,
+ int allow_update)
+{
+ SYSNO sysno0 = 0;
+ RecordAttr *recordAttr;
+ struct recExtractCtrl extractCtrl;
+ int r;
+ const char *matchStr = 0;
+ RecType recType = NULL;
+ void *clientData;
+ Record rec;
+ long recordOffset = 0;
+ struct zebra_fetch_control fc;
+ const char *pr_fname = fname; /* filename to print .. */
+ int show_progress = zh->records_processed < zh->m_file_verbose_limit ? 1:0;
+
+ if (!pr_fname)
+ pr_fname = "<no file>"; /* make it printable if file is omitted */
+
+ fc.fd = -1;
+ fc.record_int_buf = buf;
+ fc.record_int_len = buf_size;
+ fc.record_int_pos = 0;
+ fc.offset_end = 0;
+ fc.record_offset = 0;
+
+ extractCtrl.offset = 0;
+ extractCtrl.readf = zebra_record_int_read;
+ extractCtrl.seekf = zebra_record_int_seek;
+ extractCtrl.tellf = zebra_record_int_tell;
+ extractCtrl.endf = zebra_record_int_end;
+ extractCtrl.first_record = 1;
+ extractCtrl.fh = &fc;
+
+ zebra_rec_keys_reset(zh->reg->keys);
+
+#if NATTR
+ zebra_rec_keys_reset(zh->reg->sortKeys);
+#else
+ zh->reg->sortKeys.buf_used = 0;
+#endif
+ if (zebraExplain_curDatabase (zh->reg->zei, zh->basenames[0]))
+ {
+ if (zebraExplain_newDatabase (zh->reg->zei, zh->basenames[0],
+ zh->m_explain_database))
+ return ZEBRA_FAIL;
+ }
+
+ if (recordType && *recordType)
+ {
+ yaz_log (YLOG_DEBUG, "Record type explicitly specified: %s", recordType);
+ recType = recType_byName (zh->reg->recTypes, zh->res, recordType,
+ &clientData);
+ }
+ else
+ {
+ if (!(zh->m_record_type))
+ {
+ yaz_log (YLOG_WARN, "No such record type defined");
+ return ZEBRA_FAIL;
+ }
+ yaz_log (YLOG_DEBUG, "Get record type from rgroup: %s",zh->m_record_type);
+ recType = recType_byName (zh->reg->recTypes, zh->res,
+ zh->m_record_type, &clientData);
+ recordType = zh->m_record_type;
+ }
+
+ if (!recType)
+ {
+ yaz_log (YLOG_WARN, "No such record type: %s", zh->m_record_type);
+ return ZEBRA_FAIL;
+ }
+
+ extractCtrl.init = extract_init;
+ extractCtrl.tokenAdd = extract_token_add;
+ extractCtrl.schemaAdd = extract_schema_add;
+ extractCtrl.dh = zh->reg->dh;
+ extractCtrl.handle = zh;
+ extractCtrl.match_criteria[0] = '\0';
+ extractCtrl.staticrank = 0;
+
+ init_extractCtrl(zh, &extractCtrl);
+
+ extract_set_store_data_prepare(&extractCtrl);
+
+ r = (*recType->extract)(clientData, &extractCtrl);
+
+ if (r == RECCTRL_EXTRACT_EOF)
+ return ZEBRA_FAIL;
+ else if (r == RECCTRL_EXTRACT_ERROR_GENERIC)
+ {
+ /* error occured during extraction ... */
+ yaz_log (YLOG_WARN, "extract error: generic");
+ return ZEBRA_FAIL;
+ }
+ else if (r == RECCTRL_EXTRACT_ERROR_NO_SUCH_FILTER)
+ {
+ /* error occured during extraction ... */
+ yaz_log (YLOG_WARN, "extract error: no such filter");
+ return ZEBRA_FAIL;
+ }
+
+ if (extractCtrl.match_criteria[0])
+ match_criteria = extractCtrl.match_criteria;
+
+ if (!sysno) {
+
+ sysno = &sysno0;
+
+ if (match_criteria && *match_criteria) {
+ matchStr = match_criteria;
+ } else {
+ if (zh->m_record_id && *zh->m_record_id) {
+ matchStr = fileMatchStr (zh, zh->reg->keys, pr_fname,
+ zh->m_record_id);
+ if (!matchStr)
+ {
+ yaz_log (YLOG_WARN, "Bad match criteria (recordID)");
+ return ZEBRA_FAIL;
+ }
+ }
+ }
+ if (matchStr)
+ {
+ int db_ord = zebraExplain_get_database_ord(zh->reg->zei);
+ char *rinfo = dict_lookup_ord(zh->reg->matchDict, db_ord,
+ matchStr);
+ if (rinfo)
+ {
+ assert(*rinfo == sizeof(*sysno));
+ memcpy (sysno, rinfo+1, sizeof(*sysno));
+ }
+ }
+ }
+ if (zebra_rec_keys_empty(zh->reg->keys))
+ {
+ /* the extraction process returned no information - the record
+ is probably empty - unless flagShowRecords is in use */
+ if (test_mode)
+ return ZEBRA_OK;
+ }
+
+ if (! *sysno)
+ {
+ /* new record */
+ if (delete_flag)
+ {
+ yaz_log (YLOG_LOG, "delete %s %s %ld", recordType,
+ pr_fname, (long) recordOffset);
+ yaz_log (YLOG_WARN, "cannot delete record above (seems new)");
+ return ZEBRA_FAIL;
+ }
+ if (show_progress)
+ yaz_log (YLOG_LOG, "add %s %s %ld", recordType, pr_fname,
+ (long) recordOffset);
+ rec = rec_new (zh->reg->records);
+
+ *sysno = rec->sysno;
+
+ recordAttr = rec_init_attr (zh->reg->zei, rec);
+ recordAttr->staticrank = extractCtrl.staticrank;
+
+ if (matchStr)
+ {
+ int db_ord = zebraExplain_get_database_ord(zh->reg->zei);
+ dict_insert_ord(zh->reg->matchDict, db_ord, matchStr,
+ sizeof(*sysno), sysno);
+ }
+#if NATTR
+ extract_flushSortKeys (zh, *sysno, 1, zh->reg->sortKeys);
+#else
+ extract_flushSortKeys (zh, *sysno, 1, &zh->reg->sortKeys);
+#endif
+
+#if 0
+ print_rec_keys(zh, zh->reg->keys);
+#endif
+ extract_flushRecordKeys (zh, *sysno, 1, zh->reg->keys,
+ recordAttr->staticrank);
+ zh->records_inserted++;
+ }
+ else
+ {
+ /* record already exists */
+ zebra_rec_keys_t delkeys = zebra_rec_keys_open();
+#if NATTR
+ zebra_rec_keys_t sortKeys = zebra_rec_keys_open();
+#else
+ struct sortKeys sortKeys;
+#endif
+
+ if (!allow_update)
+ {
+ yaz_log (YLOG_LOG, "skipped %s %s %ld",
+ recordType, pr_fname, (long) recordOffset);
+ logRecord(zh);
+ return ZEBRA_FAIL;
+ }
+
+ rec = rec_get (zh->reg->records, *sysno);
+ assert (rec);
+
+ recordAttr = rec_init_attr (zh->reg->zei, rec);
+
+ zebra_rec_keys_set_buf(delkeys,
+ rec->info[recInfo_delKeys],
+ rec->size[recInfo_delKeys],
+ 0);
+#if NATTR
+ zebra_rec_keys_set_buf(sortKeys,
+ rec->info[recInfo_sortKeys],
+ rec->size[recInfo_sortKeys],
+ 0);
+#else
+ sortKeys.buf_used = rec->size[recInfo_sortKeys];
+ sortKeys.buf = rec->info[recInfo_sortKeys];
+#endif
+
+#if NATTR
+ extract_flushSortKeys (zh, *sysno, 0, sortKeys);
+#else
+ extract_flushSortKeys (zh, *sysno, 0, &sortKeys);
+#endif
+ extract_flushRecordKeys (zh, *sysno, 0, delkeys,
+ recordAttr->staticrank);
+ if (delete_flag)
+ {
+ /* record going to be deleted */
+ if (zebra_rec_keys_empty(delkeys))
+ {
+ yaz_log (YLOG_LOG, "delete %s %s %ld", recordType,
+ pr_fname, (long) recordOffset);
+ yaz_log (YLOG_WARN, "cannot delete file above, "
+ "storeKeys false (3)");
+ }
+ else
+ {
+ if (show_progress)
+ yaz_log (YLOG_LOG, "delete %s %s %ld", recordType,
+ pr_fname, (long) recordOffset);
+ zh->records_deleted++;
+ if (matchStr)
+ {
+ int db_ord = zebraExplain_get_database_ord(zh->reg->zei);
+ dict_delete_ord(zh->reg->matchDict, db_ord, matchStr);
+ }
+ rec_del (zh->reg->records, &rec);
+ }
+ rec_rm (&rec);
+ logRecord(zh);
+ return ZEBRA_OK;
+ }
+ else
+ {
+ if (show_progress)
+ yaz_log (YLOG_LOG, "update %s %s %ld", recordType,
+ pr_fname, (long) recordOffset);
+ recordAttr->staticrank = extractCtrl.staticrank;
+#if NATTR
+ extract_flushSortKeys (zh, *sysno, 1, zh->reg->sortKeys);
+#else
+ extract_flushSortKeys (zh, *sysno, 1, &zh->reg->sortKeys);
+#endif
+ extract_flushRecordKeys (zh, *sysno, 1, zh->reg->keys,
+ recordAttr->staticrank);
+ zh->records_updated++;
+ }
+ zebra_rec_keys_close(delkeys);
+#if NATTR
+ zebra_rec_keys_close(sortKeys);
+#endif
+ }
+ /* update file type */
+ xfree (rec->info[recInfo_fileType]);
+ rec->info[recInfo_fileType] =
+ rec_strdup (recordType, &rec->size[recInfo_fileType]);
+
+ /* update filename */
+ xfree (rec->info[recInfo_filename]);
+ rec->info[recInfo_filename] =
+ rec_strdup (fname, &rec->size[recInfo_filename]);
+
+ /* update delete keys */
+ xfree (rec->info[recInfo_delKeys]);
+ if (!zebra_rec_keys_empty(zh->reg->keys) && zh->m_store_keys == 1)
+ {
+ zebra_rec_keys_get_buf(zh->reg->keys,
+ &rec->info[recInfo_delKeys],
+ &rec->size[recInfo_delKeys]);
+ }
+ else
+ {
+ rec->info[recInfo_delKeys] = NULL;
+ rec->size[recInfo_delKeys] = 0;
+ }
+ /* update sort keys */
+ xfree (rec->info[recInfo_sortKeys]);