\brief indexes records and extract tokens for indexing and sorting
*/
+#if HAVE_CONFIG_H
+#include <config.h>
+#endif
#include <stdio.h>
#include <assert.h>
#include <ctype.h>
zebra_map_t zm)
{
struct snip_rec_info *h = p->extractCtrl->handle;
-
- const char *b = p->term_buf;
- char buf[IT_MAX_WORD+1];
- const char **map = 0;
- int i = 0, remain = p->term_len;
- const char *start = b;
- const char *last = 0;
-
- if (remain > 0)
- map = zebra_maps_input(zm, &b, remain, 1);
-
- while (remain > 0 && i < IT_MAX_WORD)
- {
- while (map && *map && **map == *CHR_SPACE)
- {
- remain = p->term_len - (b - p->term_buf);
-
- if (i == 0)
- start = b; /* set to first non-ws area */
- if (remain > 0)
- {
- int first = i ? 0 : 1; /* first position */
-
- map = zebra_maps_input(zm, &b, remain, first);
- }
- else
- map = 0;
- }
- if (!map)
- break;
-
- if (i && i < IT_MAX_WORD)
- buf[i++] = *CHR_SPACE;
- while (map && *map && **map != *CHR_SPACE)
- {
- const char *cp = *map;
-
- if (**map == *CHR_CUT)
- {
- i = 0;
- }
- else
- {
- if (i >= IT_MAX_WORD)
- break;
- while (i < IT_MAX_WORD && *cp)
- buf[i++] = *(cp++);
- }
- last = b;
- remain = p->term_len - (b - p->term_buf);
- if (remain > 0)
- {
- map = zebra_maps_input(zm, &b, remain, 0);
- }
- else
- map = 0;
- }
- }
- if (!i)
- return;
- if (last && start != last && zebra_maps_is_index(zm))
+ if (p->term_len && p->term_buf && zebra_maps_is_index(zm))
zebra_snippets_appendn(h->snippets, p->seqno, 0, ord,
- start, last - start);
+ p->term_buf, p->term_len);
+ p->seqno++;
}
static void snippet_add_incomplete_field(RecWord *p, int ord, zebra_map_t zm)
while (map)
{
- char buf[IT_MAX_WORD+1];
- int i, remain;
+ int remain;
/* Skip spaces */
while (map && *map && **map == *CHR_SPACE)
{
zebra_snippets_appendn(h->snippets, p->seqno, 1, ord,
start, last - start);
-
}
start = last;
-
- i = 0;
while (map && *map && **map != *CHR_SPACE)
{
- const char *cp = *map;
-
- while (i < IT_MAX_WORD && *cp)
- buf[i++] = *(cp++);
remain = p->term_len - (b - p->term_buf);
last = b;
if (remain > 0)
else
map = 0;
}
- if (!i)
- return;
+ if (start == last)
+ return ;
if (first)
{
{
struct recExtractCtrl extractCtrl;
struct snip_rec_info info;
- int r;
extractCtrl.stream = stream;
extractCtrl.first_record = 1;
extractCtrl.setStoreData = 0;
- r = (*rt->extract)(recTypeClientData, &extractCtrl);
-
+ (*rt->extract)(recTypeClientData, &extractCtrl);
}
static void searchRecordKey(ZebraHandle zh,
char gprefix[128];
char ext[128];
char ext_res[128];
- struct file_read_info *fi = 0;
const char *original_record_type = 0;
RecType recType;
void *recTypeClientData;
if (sysno && (action == action_delete || action == action_a_delete))
{
streamp = 0;
- fi = 0;
}
else
{
}
}
-static void extract_flush_record_keys(
- ZebraHandle zh, zint sysno, int cmd,
- zebra_rec_keys_t reckeys,
- zint staticrank)
-{
- ZebraExplainInfo zei = zh->reg->zei;
-
- extract_rec_keys_adjust(zh, cmd, reckeys);
-
- if (log_level_details)
- {
- yaz_log(log_level_details, "Keys for record " ZINT_FORMAT " %s",
- sysno, cmd ? "insert" : "delete");
- extract_rec_keys_log(zh, cmd, reckeys, log_level_details);
- }
-
- if (!zh->reg->key_block)
- {
- int mem = 1024*1024 * atoi( res_get_def( zh->res, "memmax", "8"));
- const char *key_tmp_dir = res_get_def(zh->res, "keyTmpDir", ".");
- int use_threads = atoi(res_get_def(zh->res, "threads", "1"));
- zh->reg->key_block = key_block_create(mem, key_tmp_dir, use_threads);
- }
- zebraExplain_recordCountIncrement(zei, cmd ? 1 : -1);
-
-#if 0
- yaz_log(YLOG_LOG, "sysno=" ZINT_FORMAT " cmd=%d", sysno, cmd);
- print_rec_keys(zh, reckeys);
-#endif
- if (zebra_rec_keys_rewind(reckeys))
- {
- size_t slen;
- const char *str;
- struct it_key key_in;
- while(zebra_rec_keys_read(reckeys, &str, &slen, &key_in))
- {
- key_block_write(zh->reg->key_block, sysno,
- &key_in, cmd, str, slen,
- staticrank, zh->m_staticrank);
- }
- }
-}
-
+#if FLUSH2
static void extract_flush_record_keys2(
ZebraHandle zh, zint sysno,
zebra_rec_keys_t ins_keys, zint ins_rank,
}
yaz_log(log_level_extract, "normal=%d optimized=%d", normal, optimized);
}
+#else
+static void extract_flush_record_keys(
+ ZebraHandle zh, zint sysno, int cmd,
+ zebra_rec_keys_t reckeys,
+ zint staticrank)
+{
+ ZebraExplainInfo zei = zh->reg->zei;
+
+ extract_rec_keys_adjust(zh, cmd, reckeys);
+
+ if (log_level_details)
+ {
+ yaz_log(log_level_details, "Keys for record " ZINT_FORMAT " %s",
+ sysno, cmd ? "insert" : "delete");
+ extract_rec_keys_log(zh, cmd, reckeys, log_level_details);
+ }
+ if (!zh->reg->key_block)
+ {
+ int mem = 1024*1024 * atoi( res_get_def( zh->res, "memmax", "8"));
+ const char *key_tmp_dir = res_get_def(zh->res, "keyTmpDir", ".");
+ int use_threads = atoi(res_get_def(zh->res, "threads", "1"));
+ zh->reg->key_block = key_block_create(mem, key_tmp_dir, use_threads);
+ }
+ zebraExplain_recordCountIncrement(zei, cmd ? 1 : -1);
+
+#if 0
+ yaz_log(YLOG_LOG, "sysno=" ZINT_FORMAT " cmd=%d", sysno, cmd);
+ print_rec_keys(zh, reckeys);
+#endif
+ if (zebra_rec_keys_rewind(reckeys))
+ {
+ size_t slen;
+ const char *str;
+ struct it_key key_in;
+ while(zebra_rec_keys_read(reckeys, &str, &slen, &key_in))
+ {
+ key_block_write(zh->reg->key_block, sysno,
+ &key_in, cmd, str, slen,
+ staticrank, zh->m_staticrank);
+ }
+ }
+}
+#endif
ZEBRA_RES zebra_rec_keys_to_snippets1(ZebraHandle zh,
zebra_rec_keys_t reckeys,
if (!i)
return;
extract_add_string(p, zm, buf, i);
+ p->seqno++;
}
static void extract_add_icu(RecWord *p, zebra_map_t zm)