-/* $Id: extract.c,v 1.278 2008-01-26 15:32:51 adam Exp $
- Copyright (C) 1995-2007
- Index Data ApS
-
-This file is part of the Zebra server.
+/* This file is part of the Zebra server.
+ Copyright (C) 1994-2011 Index Data
Zebra is free software; you can redistribute it and/or modify it under
the terms of the GNU General Public License as published by the Free
\brief indexes records and extract tokens for indexing and sorting
*/
+#if HAVE_CONFIG_H
+#include <config.h>
+#endif
#include <stdio.h>
#include <assert.h>
#include <ctype.h>
/* eventually this the 0-case code will be removed */
#define FLUSH2 1
-void extract_flush_record_keys2(ZebraHandle zh, zint sysno,
- zebra_rec_keys_t ins_keys,
- zint ins_rank,
- zebra_rec_keys_t del_keys,
- zint del_rank);
+#if FLUSH2
+static void extract_flush_record_keys2(ZebraHandle zh, zint sysno,
+ zebra_rec_keys_t ins_keys,
+ zint ins_rank,
+ zebra_rec_keys_t del_keys,
+ zint del_rank);
+#else
+static void extract_flush_record_keys(ZebraHandle zh, zint sysno,
+ int cmd,
+ zebra_rec_keys_t reckeys,
+ zint staticrank);
+#endif
static void zebra_init_log_level(void)
{
}
}
+static WRBUF wrbuf_hex_str(const char *cstr)
+{
+ size_t i;
+ WRBUF w = wrbuf_alloc();
+ for (i = 0; cstr[i]; i++)
+ {
+ if (cstr[i] < ' ' || cstr[i] > 126)
+ wrbuf_printf(w, "\\%02X", cstr[i] & 0xff);
+ else
+ wrbuf_putc(w, cstr[i]);
+ }
+ return w;
+}
+
+
static void extract_flush_sort_keys(ZebraHandle zh, zint sysno,
int cmd, zebra_rec_keys_t skp);
static void extract_schema_add(struct recExtractCtrl *p, Odr_oid *oid);
zebra_map_t zm)
{
struct snip_rec_info *h = p->extractCtrl->handle;
-
- const char *b = p->term_buf;
- char buf[IT_MAX_WORD+1];
- const char **map = 0;
- int i = 0, remain = p->term_len;
- const char *start = b;
- const char *last = 0;
-
- if (remain > 0)
- map = zebra_maps_input(zm, &b, remain, 1);
-
- while (remain > 0 && i < IT_MAX_WORD)
- {
- while (map && *map && **map == *CHR_SPACE)
- {
- remain = p->term_len - (b - p->term_buf);
-
- if (i == 0)
- start = b; /* set to first non-ws area */
- if (remain > 0)
- {
- int first = i ? 0 : 1; /* first position */
-
- map = zebra_maps_input(zm, &b, remain, first);
- }
- else
- map = 0;
- }
- if (!map)
- break;
-
- if (i && i < IT_MAX_WORD)
- buf[i++] = *CHR_SPACE;
- while (map && *map && **map != *CHR_SPACE)
- {
- const char *cp = *map;
-
- if (**map == *CHR_CUT)
- {
- i = 0;
- }
- else
- {
- if (i >= IT_MAX_WORD)
- break;
- while (i < IT_MAX_WORD && *cp)
- buf[i++] = *(cp++);
- }
- last = b;
- remain = p->term_len - (b - p->term_buf);
- if (remain > 0)
- {
- map = zebra_maps_input(zm, &b, remain, 0);
- }
- else
- map = 0;
- }
- }
- if (!i)
- return;
- if (last && start != last && zebra_maps_is_index(zm))
+ if (p->term_len && p->term_buf && zebra_maps_is_index(zm))
zebra_snippets_appendn(h->snippets, p->seqno, 0, ord,
- start, last - start);
+ p->term_buf, p->term_len);
+ p->seqno++;
}
static void snippet_add_incomplete_field(RecWord *p, int ord, zebra_map_t zm)
while (map)
{
- char buf[IT_MAX_WORD+1];
- int i, remain;
+ int remain;
/* Skip spaces */
while (map && *map && **map == *CHR_SPACE)
{
zebra_snippets_appendn(h->snippets, p->seqno, 1, ord,
start, last - start);
-
}
start = last;
-
- i = 0;
while (map && *map && **map != *CHR_SPACE)
{
- const char *cp = *map;
-
- while (i < IT_MAX_WORD && *cp)
- buf[i++] = *(cp++);
remain = p->term_len - (b - p->term_buf);
last = b;
if (remain > 0)
else
map = 0;
}
- if (!i)
- return;
+ if (start == last)
+ return ;
if (first)
{
{
struct snip_rec_info *h = p->extractCtrl->handle;
ZebraHandle zh = h->zh;
- zebra_map_t zm = zebra_map_get(zh->reg->zebra_maps, p->index_type);
+ zebra_map_t zm = zebra_map_get_or_add(zh->reg->zebra_maps, p->index_type);
if (zm)
{
{
struct recExtractCtrl extractCtrl;
struct snip_rec_info info;
- int r;
extractCtrl.stream = stream;
extractCtrl.first_record = 1;
extractCtrl.setStoreData = 0;
- r = (*rt->extract)(recTypeClientData, &extractCtrl);
-
+ (*rt->extract)(recTypeClientData, &extractCtrl);
}
static void searchRecordKey(ZebraHandle zh,
attname_str[i++] = *s;
attname_str[i] = '\0';
}
-
- searchRecordKey(zh, reckeys, attname_str, ws, 32);
-
if (*s != ')')
{
yaz_log(YLOG_WARN, "Missing ) in match criteria %s in group %s",
}
s++;
+ searchRecordKey(zh, reckeys, attname_str, ws, 32);
+ if (0) /* for debugging */
+ {
+ for (i = 0; i<32; i++)
+ {
+ if (ws[i])
+ {
+ WRBUF w = wrbuf_hex_str(ws[i]);
+ yaz_log(YLOG_LOG, "ws[%d] = %s", i, wrbuf_cstr(w));
+ wrbuf_destroy(w);
+ }
+ }
+ }
+
for (i = 0; i<32; i++)
if (ws[i])
{
return NULL;
}
*dst = '\0';
+
+ if (0) /* for debugging */
+ {
+ WRBUF w = wrbuf_hex_str(dstBuf);
+ yaz_log(YLOG_LOG, "get_match_from_spec %s", wrbuf_cstr(w));
+ wrbuf_destroy(w);
+ }
+
return dstBuf;
}
"", 0);
}
+/* forward declaration */
ZEBRA_RES zebra_extract_records_stream(ZebraHandle zh,
struct ZebraRecStream *stream,
enum zebra_recctrl_action_t action,
- int test_mode,
const char *recordType,
zint *sysno,
const char *match_criteria,
char gprefix[128];
char ext[128];
char ext_res[128];
- struct file_read_info *fi = 0;
const char *original_record_type = 0;
RecType recType;
void *recTypeClientData;
if (sysno && (action == action_delete || action == action_a_delete))
{
streamp = 0;
- fi = 0;
}
else
{
}
r = zebra_extract_records_stream(zh, streamp,
action,
- 0, /* tst_mode */
zh->m_record_type,
sysno,
0, /*match_criteria */
ZEBRA_RES zebra_buffer_extract_record(ZebraHandle zh,
const char *buf, size_t buf_size,
enum zebra_recctrl_action_t action,
- int test_mode,
const char *recordType,
zint *sysno,
const char *match_criteria,
res = zebra_extract_records_stream(zh, &stream,
action,
- test_mode,
recordType,
sysno,
match_criteria,
return res;
}
-ZEBRA_RES zebra_extract_records_stream(ZebraHandle zh,
- struct ZebraRecStream *stream,
- enum zebra_recctrl_action_t action,
- int test_mode,
- const char *recordType,
- zint *sysno,
- const char *match_criteria,
- const char *fname,
- RecType recType,
- void *recTypeClientData)
-{
- ZEBRA_RES res = ZEBRA_OK;
- while (1)
- {
- int more = 0;
- res = zebra_extract_record_stream(zh, stream,
- action,
- test_mode,
- recordType,
- sysno,
- match_criteria,
- fname,
- recType, recTypeClientData, &more);
- if (!more)
- {
- res = ZEBRA_OK;
- break;
- }
- if (res != ZEBRA_OK)
- break;
- if (sysno)
- break;
- }
- return res;
-}
-
-
-static WRBUF wrbuf_hex_str(const char *cstr)
-{
- size_t i;
- WRBUF w = wrbuf_alloc();
- for (i = 0; cstr[i]; i++)
- {
- if (cstr[i] < ' ' || cstr[i] > 126)
- wrbuf_printf(w, "\\%02X", cstr[i] & 0xff);
- else
- wrbuf_putc(w, cstr[i]);
- }
- return w;
-}
-
-ZEBRA_RES zebra_extract_record_stream(ZebraHandle zh,
- struct ZebraRecStream *stream,
- enum zebra_recctrl_action_t action,
- int test_mode,
- const char *recordType,
- zint *sysno,
- const char *match_criteria,
- const char *fname,
- RecType recType,
- void *recTypeClientData,
- int *more)
-
+static ZEBRA_RES zebra_extract_record_stream(ZebraHandle zh,
+ struct ZebraRecStream *stream,
+ enum zebra_recctrl_action_t action,
+ const char *recordType,
+ zint *sysno,
+ const char *match_criteria,
+ const char *fname,
+ RecType recType,
+ void *recTypeClientData,
+ int *more)
+
{
zint sysno0 = 0;
RecordAttr *recordAttr;
}
*more = 1;
+
+ if (zh->m_flag_rw == 0)
+ {
+ yaz_log(YLOG_LOG, "test %s %s " ZINT_FORMAT, recordType,
+ pr_fname, (zint) start_offset);
+ /* test mode .. Do not perform match */
+ return ZEBRA_OK;
+ }
+
if (!sysno)
{
sysno = &sysno0;
-
- if (match_criteria && *match_criteria) {
+
+ if (match_criteria && *match_criteria)
matchStr = match_criteria;
- } else {
- if (zh->m_record_id && *zh->m_record_id) {
+ else
+ {
+ if (zh->m_record_id && *zh->m_record_id)
+ {
matchStr = get_match_from_spec(zh, zh->reg->keys, pr_fname,
zh->m_record_id);
if (!matchStr)
pr_fname, (zint) start_offset);
return ZEBRA_FAIL;
}
+ if (0 && matchStr)
+ {
+ WRBUF w = wrbuf_alloc();
+ size_t i;
+ for (i = 0; i < strlen(matchStr); i++)
+ {
+ wrbuf_printf(w, "%02X", matchStr[i] & 0xff);
+ }
+ yaz_log(YLOG_LOG, "Got match %s", wrbuf_cstr(w));
+ wrbuf_destroy(w);
+ }
}
}
if (matchStr)
}
}
- if (zebra_rec_keys_empty(zh->reg->keys))
- {
- /* the extraction process returned no information - the record
- is probably empty - unless flagShowRecords is in use */
- if (test_mode)
- return ZEBRA_OK;
- }
-
if (! *sysno)
{
/* new record AKA does not exist already */
if (action == action_delete)
{
- yaz_log(YLOG_LOG, "delete %s %s " ZINT_FORMAT, recordType,
- pr_fname, (zint) start_offset);
+ yaz_log(YLOG_LOG, "delete %s %s " ZINT_FORMAT, recordType,
+ pr_fname, (zint) start_offset);
yaz_log(YLOG_WARN, "cannot delete record above (seems new)");
return ZEBRA_FAIL;
}
return ZEBRA_OK;
}
+/** \brief extracts records from stream
+ \param zh Zebra Handle
+ \param stream stream that we read from
+ \param action (action_insert, action_replace, action_delete, ..)
+ \param recordType Record filter type "grs.xml", etc.
+ \param sysno pointer to sysno if already known; NULL otherwise
+ \param match_criteria (NULL if not already given)
+ \param fname filename that we read from (for logging purposes only)
+ \param recType record type
+ \param recTypeClientData client data for record type
+ \returns ZEBRA_OK for success; ZEBRA_FAIL for failure
+*/
+ZEBRA_RES zebra_extract_records_stream(ZebraHandle zh,
+ struct ZebraRecStream *stream,
+ enum zebra_recctrl_action_t action,
+ const char *recordType,
+ zint *sysno,
+ const char *match_criteria,
+ const char *fname,
+ RecType recType,
+ void *recTypeClientData)
+{
+ ZEBRA_RES res = ZEBRA_OK;
+ while (1)
+ {
+ int more = 0;
+ res = zebra_extract_record_stream(zh, stream,
+ action,
+ recordType,
+ sysno,
+ match_criteria,
+ fname,
+ recType, recTypeClientData, &more);
+ if (!more)
+ {
+ res = ZEBRA_OK;
+ break;
+ }
+ if (res != ZEBRA_OK)
+ break;
+ if (sysno)
+ break;
+ }
+ return res;
+}
+
ZEBRA_RES zebra_extract_explain(void *handle, Record rec, data1_node *n)
{
ZebraHandle zh = (ZebraHandle) handle;
return ZEBRA_OK;
}
+void zebra_it_key_str_dump(ZebraHandle zh, struct it_key *key,
+ const char *str, size_t slen, NMEM nmem, int level)
+{
+ char keystr[200]; /* room for zints to print */
+ char *dst_term = 0;
+ int ord = CAST_ZINT_TO_INT(key->mem[0]);
+ const char *index_type;
+ int i;
+ const char *string_index;
+
+ zebraExplain_lookup_ord(zh->reg->zei, ord, &index_type,
+ 0/* db */, &string_index);
+ assert(index_type);
+ zebra_term_untrans_iconv(zh, nmem, index_type,
+ &dst_term, str);
+ *keystr = '\0';
+ for (i = 0; i < key->len; i++)
+ {
+ sprintf(keystr + strlen(keystr), ZINT_FORMAT " ", key->mem[i]);
+ }
+
+ if (*str < CHR_BASE_CHAR)
+ {
+ int i;
+ char dst_buf[200]; /* room for special chars */
+
+ strcpy(dst_buf , "?");
+
+ if (!strcmp(str, ""))
+ strcpy(dst_buf, "alwaysmatches");
+ if (!strcmp(str, FIRST_IN_FIELD_STR))
+ strcpy(dst_buf, "firstinfield");
+ else if (!strcmp(str, CHR_UNKNOWN))
+ strcpy(dst_buf, "unknown");
+ else if (!strcmp(str, CHR_SPACE))
+ strcpy(dst_buf, "space");
+
+ for (i = 0; i<slen; i++)
+ {
+ sprintf(dst_buf + strlen(dst_buf), " %d", str[i] & 0xff);
+ }
+ yaz_log(level, "%s%s %s %s", keystr, index_type,
+ string_index, dst_buf);
+
+ }
+ else
+ yaz_log(level, "%s%s %s \"%s\"", keystr, index_type,
+ string_index, dst_term);
+}
+
void extract_rec_keys_log(ZebraHandle zh, int is_insert,
zebra_rec_keys_t reckeys,
int level)
while(zebra_rec_keys_read(reckeys, &str, &slen, &key))
{
- char keystr[200]; /* room for zints to print */
- char *dst_term = 0;
- int ord = CAST_ZINT_TO_INT(key.mem[0]);
- const char *index_type;
- int i;
- const char *string_index;
-
- zebraExplain_lookup_ord(zh->reg->zei, ord, &index_type,
- 0/* db */, &string_index);
- assert(index_type);
- zebra_term_untrans_iconv(zh, nmem, index_type,
- &dst_term, str);
- *keystr = '\0';
- for (i = 0; i<key.len; i++)
- {
- sprintf(keystr + strlen(keystr), ZINT_FORMAT " ", key.mem[i]);
- }
-
- if (*str < CHR_BASE_CHAR)
- {
- int i;
- char dst_buf[200]; /* room for special chars */
-
- strcpy(dst_buf , "?");
-
- if (!strcmp(str, ""))
- strcpy(dst_buf, "alwaysmatches");
- if (!strcmp(str, FIRST_IN_FIELD_STR))
- strcpy(dst_buf, "firstinfield");
- else if (!strcmp(str, CHR_UNKNOWN))
- strcpy(dst_buf, "unknown");
- else if (!strcmp(str, CHR_SPACE))
- strcpy(dst_buf, "space");
-
- for (i = 0; i<slen; i++)
- {
- sprintf(dst_buf + strlen(dst_buf), " %d", str[i] & 0xff);
- }
- yaz_log(level, "%s%s %s %s", keystr, index_type,
- string_index, dst_buf);
-
- }
- else
- yaz_log(level, "%s%s %s \"%s\"", keystr, index_type,
- string_index, dst_term);
-
+ zebra_it_key_str_dump(zh, &key, str, slen, nmem, level);
nmem_reset(nmem);
}
nmem_destroy(nmem);
}
}
-void extract_flush_record_keys2(ZebraHandle zh, zint sysno,
- zebra_rec_keys_t ins_keys, zint ins_rank,
- zebra_rec_keys_t del_keys, zint del_rank)
+#if FLUSH2
+static void extract_flush_record_keys2(
+ ZebraHandle zh, zint sysno,
+ zebra_rec_keys_t ins_keys, zint ins_rank,
+ zebra_rec_keys_t del_keys, zint del_rank)
{
ZebraExplainInfo zei = zh->reg->zei;
int normal = 0;
}
yaz_log(log_level_extract, "normal=%d optimized=%d", normal, optimized);
}
+#else
+static void extract_flush_record_keys(
+ ZebraHandle zh, zint sysno, int cmd,
+ zebra_rec_keys_t reckeys,
+ zint staticrank)
+{
+ ZebraExplainInfo zei = zh->reg->zei;
+ extract_rec_keys_adjust(zh, cmd, reckeys);
+
+ if (log_level_details)
+ {
+ yaz_log(log_level_details, "Keys for record " ZINT_FORMAT " %s",
+ sysno, cmd ? "insert" : "delete");
+ extract_rec_keys_log(zh, cmd, reckeys, log_level_details);
+ }
+
+ if (!zh->reg->key_block)
+ {
+ int mem = 1024*1024 * atoi( res_get_def( zh->res, "memmax", "8"));
+ const char *key_tmp_dir = res_get_def(zh->res, "keyTmpDir", ".");
+ int use_threads = atoi(res_get_def(zh->res, "threads", "1"));
+ zh->reg->key_block = key_block_create(mem, key_tmp_dir, use_threads);
+ }
+ zebraExplain_recordCountIncrement(zei, cmd ? 1 : -1);
+
+#if 0
+ yaz_log(YLOG_LOG, "sysno=" ZINT_FORMAT " cmd=%d", sysno, cmd);
+ print_rec_keys(zh, reckeys);
+#endif
+ if (zebra_rec_keys_rewind(reckeys))
+ {
+ size_t slen;
+ const char *str;
+ struct it_key key_in;
+ while(zebra_rec_keys_read(reckeys, &str, &slen, &key_in))
+ {
+ key_block_write(zh->reg->key_block, sysno,
+ &key_in, cmd, str, slen,
+ staticrank, zh->m_staticrank);
+ }
+ }
+}
+#endif
ZEBRA_RES zebra_rec_keys_to_snippets1(ZebraHandle zh,
zebra_rec_keys_t reckeys,
ch = zebraExplain_lookup_attr_str(zei, cat, p->index_type, p->index_name);
if (ch < 0)
ch = zebraExplain_add_attr_str(zei, cat, p->index_type, p->index_name);
- key.len = 2;
+ key.len = 3;
key.mem[0] = ch;
key.mem[1] = p->record_id;
+ key.mem[2] = p->section_id;
zebra_rec_keys_write(zh->reg->sortKeys, str, length, &key);
}
if (!i)
return;
extract_add_string(p, zm, buf, i);
+ p->seqno++;
}
static void extract_add_icu(RecWord *p, zebra_map_t zm)
zebra_map_tokenize_start(zm, p->term_buf, p->term_len);
while (zebra_map_tokenize_next(zm, &res_buf, &res_len, 0, 0))
{
+ if (res_len > IT_MAX_WORD)
+ {
+ yaz_log(YLOG_LOG, "Truncating long term %ld", (long) res_len);
+ res_len = IT_MAX_WORD;
+ }
extract_add_string(p, zm, res_buf, res_len);
p->seqno++;
}
{
ZebraHandle zh = p->extractCtrl->handle;
zebra_map_t zm = zebra_map_get_or_add(zh->reg->zebra_maps, p->index_type);
- WRBUF wrbuf;
if (log_level_details)
{
p->index_type, p->index_name,
p->seqno, p->term_len, p->term_buf);
}
- if ((wrbuf = zebra_replace(zm, 0, p->term_buf, p->term_len)))
- {
- p->term_buf = wrbuf_buf(wrbuf);
- p->term_len = wrbuf_len(wrbuf);
- }
if (zebra_maps_is_icu(zm))
{
extract_add_icu(p, zm);
const char *str;
struct it_key key_in;
- zebra_sort_sysno(si, sysno);
+ NMEM nmem = nmem_create();
+ struct sort_add_ent {
+ int ord;
+ int cmd;
+ struct sort_add_ent *next;
+ WRBUF wrbuf;
+ zint sysno;
+ zint section_id;
+ };
+ struct sort_add_ent *sort_ent_list = 0;
while (zebra_rec_keys_read(reckeys, &str, &slen, &key_in))
{
int ord = CAST_ZINT_TO_INT(key_in.mem[0]);
+ zint filter_sysno = key_in.mem[1];
+ zint section_id = key_in.mem[2];
+
+ struct sort_add_ent **e = &sort_ent_list;
+ for (; *e; e = &(*e)->next)
+ if ((*e)->ord == ord && section_id == (*e)->section_id)
+ break;
+ if (!*e)
+ {
+ *e = nmem_malloc(nmem, sizeof(**e));
+ (*e)->next = 0;
+ (*e)->wrbuf = wrbuf_alloc();
+ (*e)->ord = ord;
+ (*e)->cmd = cmd;
+ (*e)->sysno = filter_sysno ? filter_sysno : sysno;
+ (*e)->section_id = section_id;
+ }
- zebra_sort_type(si, ord);
- if (cmd == 1)
- zebra_sort_add(si, str, slen);
- else
- zebra_sort_delete(si);
+ wrbuf_write((*e)->wrbuf, str, slen);
+ wrbuf_putc((*e)->wrbuf, '\0');
+ }
+ if (sort_ent_list)
+ {
+ zint last_sysno = 0;
+ struct sort_add_ent *e = sort_ent_list;
+ for (; e; e = e->next)
+ {
+ if (last_sysno != e->sysno)
+ {
+ zebra_sort_sysno(si, e->sysno);
+ last_sysno = e->sysno;
+ }
+ zebra_sort_type(si, e->ord);
+ if (e->cmd == 1)
+ zebra_sort_add(si, e->section_id, e->wrbuf);
+ else
+ zebra_sort_delete(si, e->section_id);
+ wrbuf_destroy(e->wrbuf);
+ }
}
+ nmem_destroy(nmem);
}
}
/*
* Local variables:
* c-basic-offset: 4
+ * c-file-style: "Stroustrup"
* indent-tabs-mode: nil
* End:
* vim: shiftwidth=4 tabstop=8 expandtab