X-Git-Url: http://git.indexdata.com/?a=blobdiff_plain;f=src%2Flogic.c;h=a19be1fadfaef3ecba6ed5adaa72ca2fe55c86a6;hb=f6f232081df89d867621187764818ea777178e82;hp=e2155dd9211c1f661df19053b2a25f05728afa97;hpb=d5be779678ef159bf7f18558cf08f5a0643425a0;p=pazpar2-moved-to-github.git diff --git a/src/logic.c b/src/logic.c index e2155dd..a19be1f 100644 --- a/src/logic.c +++ b/src/logic.c @@ -1,5 +1,5 @@ /* This file is part of Pazpar2. - Copyright (C) 2006-2009 Index Data + Copyright (C) 2006-2010 Index Data Pazpar2 is free software; you can redistribute it and/or modify it under the terms of the GNU General Public License as published by the Free @@ -57,7 +57,7 @@ Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA 02110-1301 USA #include #endif - +#include "parameters.h" #include "pazpar2.h" #include "eventl.h" #include "http.h" @@ -77,12 +77,7 @@ Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA 02110-1301 USA struct parameters global_parameters = { 0, // dump_records - 0, // debug_mode - 30, // operations timeout - 60, // session timeout - 100, - 180, // Z39.50 session timeout - 15 // Connect timeout + 0 // debug_mode }; static void log_xml_doc(xmlDoc *doc) @@ -97,12 +92,22 @@ static void log_xml_doc(xmlDoc *doc) #endif if (lf && len) { - fwrite(result, 1, len, lf); + (void) fwrite(result, 1, len, lf); fprintf(lf, "\n"); } xmlFree(result); } +static void session_enter(struct session *s) +{ + yaz_mutex_enter(s->mutex); +} + +static void session_leave(struct session *s) +{ + yaz_mutex_leave(s->mutex); +} + // Recursively traverse query structure to extract terms. void pull_terms(NMEM nmem, struct ccl_rpn_node *n, char **termlist, int *num) { @@ -130,7 +135,6 @@ void pull_terms(NMEM nmem, struct ccl_rpn_node *n, char **termlist, int *num) } - static void add_facet(struct session *s, const char *type, const char *value) { int i; @@ -150,14 +154,13 @@ static void add_facet(struct session *s, const char *type, const char *value) s->termlists[i].name = nmem_strdup(s->nmem, type); s->termlists[i].termlist - = termlist_create(s->nmem, s->expected_maxrecs, - TERMLIST_HIGH_SCORE); + = termlist_create(s->nmem, TERMLIST_HIGH_SCORE); s->num_termlists = i + 1; } termlist_insert(s->termlists[i].termlist, value); } -xmlDoc *record_to_xml(struct session_database *sdb, const char *rec) +static xmlDoc *record_to_xml(struct session_database *sdb, const char *rec) { struct database *db = sdb->database; xmlDoc *rdoc = 0; @@ -184,9 +187,10 @@ xmlDoc *record_to_xml(struct session_database *sdb, const char *rec) // Add static values from session database settings if applicable static void insert_settings_parameters(struct session_database *sdb, - struct session *se, char **parms) + struct conf_service *service, + char **parms, + NMEM nmem) { - struct conf_service *service = se->service; int i; int nparms = 0; int offset = 0; @@ -197,14 +201,14 @@ static void insert_settings_parameters(struct session_database *sdb, int setting; if (md->setting == Metadata_setting_parameter && - (setting = settings_offset(service, md->name)) > 0) + (setting = settings_lookup_offset(service, md->name)) >= 0) { const char *val = session_setting_oneval(sdb, setting); if (val && nparms < MAX_XSLT_ARGS) { char *buf; int len = strlen(val); - buf = nmem_malloc(se->nmem, len + 3); + buf = nmem_malloc(nmem, len + 3); buf[0] = '\''; strcpy(buf + 1, val); buf[len+1] = '\''; @@ -230,7 +234,7 @@ static void insert_settings_values(struct session_database *sdb, xmlDoc *doc, int offset; if (md->setting == Metadata_setting_postproc && - (offset = settings_offset(service, md->name)) > 0) + (offset = settings_lookup_offset(service, md->name)) >= 0) { const char *val = session_setting_oneval(sdb, offset); if (val) @@ -244,56 +248,66 @@ static void insert_settings_values(struct session_database *sdb, xmlDoc *doc, } } -xmlDoc *normalize_record(struct session_database *sdb, struct session *se, - const char *rec) +static xmlDoc *normalize_record(struct session_database *sdb, + struct conf_service *service, + const char *rec, NMEM nmem) { - struct database_retrievalmap *m; xmlDoc *rdoc = record_to_xml(sdb, rec); + if (rdoc) { - for (m = sdb->map; m; m = m->next) + char *parms[MAX_XSLT_ARGS*2+1]; + + insert_settings_parameters(sdb, service, parms, nmem); + + if (normalize_record_transform(sdb->map, &rdoc, (const char **)parms)) { - xmlDoc *new = 0; + yaz_log(YLOG_WARN, "Normalize failed from %s", sdb->database->url); + } + else + { + insert_settings_values(sdb, rdoc, service); + if (global_parameters.dump_records) { - xmlNodePtr root = 0; - char *parms[MAX_XSLT_ARGS*2+1]; - - insert_settings_parameters(sdb, se, parms); - - new = xsltApplyStylesheet(m->stylesheet, rdoc, (const char **) parms); - root= xmlDocGetRootElement(new); - if (!new || !root || !(root->children)) - { - yaz_log(YLOG_WARN, "XSLT transformation failed from %s", - sdb->database->url); - xmlFreeDoc(new); - xmlFreeDoc(rdoc); - return 0; - } + yaz_log(YLOG_LOG, "Normalized record from %s", + sdb->database->url); + log_xml_doc(rdoc); } - - xmlFreeDoc(rdoc); - rdoc = new; } + } + return rdoc; +} - insert_settings_values(sdb, rdoc, se->service); - - if (global_parameters.dump_records) +void session_settings_dump(struct session *se, + struct session_database *db, + WRBUF w) +{ + if (db->settings) + { + int i, num = db->num_settings; + for (i = 0; i < num; i++) { - yaz_log(YLOG_LOG, "Normalized record from %s", - sdb->database->url); - log_xml_doc(rdoc); + struct setting *s = db->settings[i]; + for (;s ; s = s->next) + { + wrbuf_puts(w, "name); + wrbuf_puts(w, "\" value=\""); + wrbuf_xmlputs(w, s->value); + wrbuf_puts(w, "\"/>"); + } + if (db->settings[i]) + wrbuf_puts(w, "\n"); } } - return rdoc; } // Retrieve first defined value for 'name' for given database. // Will be extended to take into account user associated with session const char *session_setting_oneval(struct session_database *db, int offset) { - if (!db->settings[offset]) + if (offset >= db->num_settings || !db->settings[offset]) return ""; return db->settings[offset]->value; } @@ -314,9 +328,6 @@ static int prepare_map(struct session *se, struct session_database *sdb) } if ((s = session_setting_oneval(sdb, PZ_XSLT))) { - char **stylesheets; - struct database_retrievalmap **m = &sdb->map; - int num, i; char auto_stylesheet[256]; if (!strcmp(s, "auto")) @@ -341,23 +352,11 @@ static int prepare_map(struct session *se, struct session_database *sdb) yaz_log(YLOG_WARN, "No pz:requestsyntax for auto stylesheet"); } } - nmem_strsplit(se->session_nmem, ",", s, &stylesheets, &num); - for (i = 0; i < num; i++) - { - (*m) = nmem_malloc(se->session_nmem, sizeof(**m)); - (*m)->next = 0; - if (!((*m)->stylesheet = conf_load_stylesheet(stylesheets[i]))) - { - yaz_log(YLOG_FATAL|YLOG_ERRNO, "Unable to load stylesheet: %s", - stylesheets[i]); - return -1; - } - m = &(*m)->next; - } + sdb->map = normalize_cache_get(se->normalize_cache, + se->service, s); + if (!sdb->map) + return -1; } - if (!sdb->map) - yaz_log(YLOG_WARN, "No Normalization stylesheet for target %s", - sdb->database->url); return 0; } @@ -439,12 +438,12 @@ static void select_targets_callback(void *context, struct session_database *db) // Associates a set of clients with a session; // Note: Session-databases represent databases with per-session // setting overrides -int select_targets(struct session *se, struct database_criterion *crit) +static int select_targets(struct session *se, const char *filter) { while (se->clients) client_destroy(se->clients); - return session_grep_databases(se, crit, select_targets_callback); + return session_grep_databases(se, filter, select_targets_callback); } int session_active_clients(struct session *s) @@ -459,99 +458,67 @@ int session_active_clients(struct session *s) return res; } -// parses crit1=val1,crit2=val2|val3,... -static struct database_criterion *parse_filter(NMEM m, const char *buf) -{ - struct database_criterion *res = 0; - char **values; - int num; - int i; - - if (!buf || !*buf) - return 0; - nmem_strsplit(m, ",", buf, &values, &num); - for (i = 0; i < num; i++) - { - char **subvalues; - int subnum; - int subi; - struct database_criterion *new = nmem_malloc(m, sizeof(*new)); - char *eq = strchr(values[i], '='); - if (!eq) - { - yaz_log(YLOG_WARN, "Missing equal-sign in filter"); - return 0; - } - *(eq++) = '\0'; - new->name = values[i]; - nmem_strsplit(m, "|", eq, &subvalues, &subnum); - new->values = 0; - for (subi = 0; subi < subnum; subi++) - { - struct database_criterion_value *newv - = nmem_malloc(m, sizeof(*newv)); - newv->value = subvalues[subi]; - newv->next = new->values; - new->values = newv; - } - new->next = res; - res = new; - } - return res; -} enum pazpar2_error_code search(struct session *se, - char *query, char *filter, + const char *query, + const char *startrecs, const char *maxrecs, + const char *filter, const char **addinfo) { int live_channels = 0; int no_working = 0; int no_failed = 0; struct client *cl; - struct database_criterion *criteria; yaz_log(YLOG_DEBUG, "Search"); *addinfo = 0; + + session_enter(se); nmem_reset(se->nmem); se->relevance = 0; se->total_records = se->total_hits = se->total_merged = 0; + reclist_destroy(se->reclist); se->reclist = 0; se->num_termlists = 0; - criteria = parse_filter(se->nmem, filter); - live_channels = select_targets(se, criteria); - if (live_channels) + live_channels = select_targets(se, filter); + if (!live_channels) { - int maxrecs = live_channels * global_parameters.toget; // This is buggy!!! - se->reclist = reclist_create(se->nmem, maxrecs); - se->expected_maxrecs = maxrecs; - } - else + session_leave(se); return PAZPAR2_NO_TARGETS; + } + se->reclist = reclist_create(se->nmem); for (cl = se->clients; cl; cl = client_next_in_session(cl)) { + if (maxrecs) + client_set_maxrecs(cl, atoi(maxrecs)); + if (startrecs) + client_set_startrecs(cl, atoi(startrecs)); if (prepare_session_database(se, client_get_database(cl)) < 0) - { - *addinfo = client_get_database(cl)->database->url; - return PAZPAR2_CONFIG_TARGET; - } + continue; // Parse query for target if (client_parse_query(cl, query) < 0) no_failed++; else { no_working++; - if (client_prep_connection(cl)) + if (client_prep_connection(cl, se->service->z3950_operation_timeout, + se->service->z3950_session_timeout, + se->service->server->iochan_man)) client_start_search(cl); } } - - // If no queries could be mapped, we signal an error + session_leave(se); if (no_working == 0) { - *addinfo = "query"; - return PAZPAR2_MALFORMED_PARAMETER_VALUE; + if (no_failed > 0) + { + *addinfo = "query"; + return PAZPAR2_MALFORMED_PARAMETER_VALUE; + } + else + return PAZPAR2_NO_TARGETS; } return PAZPAR2_NO_ERROR; } @@ -560,22 +527,20 @@ enum pazpar2_error_code search(struct session *se, static void session_init_databases_fun(void *context, struct database *db) { struct session *se = (struct session *) context; - struct conf_service *service = se->service; struct session_database *new = nmem_malloc(se->session_nmem, sizeof(*new)); - int num = settings_num(service); int i; new->database = db; new->map = 0; - new->settings - = nmem_malloc(se->session_nmem, sizeof(struct settings *) * num); - memset(new->settings, 0, sizeof(struct settings*) * num); - - if (db->settings) + assert(db->settings); + new->settings = nmem_malloc(se->session_nmem, + sizeof(struct settings *) * db->num_settings); + new->num_settings = db->num_settings; + for (i = 0; i < db->num_settings; i++) { - for (i = 0; i < num; i++) - new->settings[i] = db->settings[i]; + struct setting *setting = db->settings[i]; + new->settings[i] = setting; } new->next = se->databases; se->databases = new; @@ -584,10 +549,7 @@ static void session_init_databases_fun(void *context, struct database *db) // Doesn't free memory associated with sdb -- nmem takes care of that static void session_database_destroy(struct session_database *sdb) { - struct database_retrievalmap *m; - - for (m = sdb->map; m; m = m->next) - xsltFreeStylesheet(m->stylesheet); + sdb->map = 0; } // Initialize session_database list -- this represents this session's view @@ -595,7 +557,7 @@ static void session_database_destroy(struct session_database *sdb) void session_init_databases(struct session *se) { se->databases = 0; - predef_grep_databases(se, se->service, 0, session_init_databases_fun); + predef_grep_databases(se, se->service, session_init_databases_fun); } // Probably session_init_databases_fun should be refactored instead of @@ -603,9 +565,12 @@ void session_init_databases(struct session *se) static struct session_database *load_session_database(struct session *se, char *id) { - struct database *db = find_database(id, 0, se->service); + struct database *db = new_database(id, se->session_nmem); + + resolve_database(se->service, db); session_init_databases_fun((void*) se, db); + // New sdb is head of se->databases list return se->databases; } @@ -629,19 +594,10 @@ void session_apply_setting(struct session *se, char *dbname, char *setting, struct session_database *sdb = find_session_database(se, dbname); struct conf_service *service = se->service; struct setting *new = nmem_malloc(se->session_nmem, sizeof(*new)); - int offset = settings_offset_cprefix(service, setting); + int offset = settings_create_offset(service, setting); - if (offset < 0) - { - yaz_log(YLOG_WARN, "Unknown setting %s", setting); - return; - } - // Jakub: This breaks the filter setting. - /*if (offset == PZ_ID) - { - yaz_log(YLOG_WARN, "No need to set pz:id setting. Ignoring"); - return; - }*/ + expand_settings_array(&sdb->settings, &sdb->num_settings, offset, + se->session_nmem); new->precedence = 0; new->target = dbname; new->name = setting; @@ -656,10 +612,6 @@ void session_apply_setting(struct session *se, char *dbname, char *setting, case PZ_XSLT: if (sdb->map) { - struct database_retrievalmap *m; - // We don't worry about the map structure -- it's in nmem - for (m = sdb->map; m; m = m->next) - xsltFreeStylesheet(m->stylesheet); sdb->map = 0; } break; @@ -674,7 +626,11 @@ void destroy_session(struct session *s) client_destroy(s->clients); for (sdb = s->databases; sdb; sdb = sdb->next) session_database_destroy(sdb); + normalize_cache_destroy(s->normalize_cache); + reclist_destroy(s->reclist); nmem_destroy(s->nmem); + service_destroy(s->service); + yaz_mutex_destroy(&s->mutex); wrbuf_destroy(s->wrbuf); } @@ -694,7 +650,6 @@ struct session *new_session(NMEM nmem, struct conf_service *service) session->num_termlists = 0; session->reclist = 0; session->clients = 0; - session->expected_maxrecs = 0; session->session_nmem = nmem; session->nmem = nmem_create(); session->wrbuf = wrbuf_alloc(); @@ -704,6 +659,10 @@ struct session *new_session(NMEM nmem, struct conf_service *service) session->watchlist[i].data = 0; session->watchlist[i].fun = 0; } + session->normalize_cache = normalize_cache_create(); + session->mutex = 0; + yaz_mutex_create(&session->mutex); + return session; } @@ -713,6 +672,7 @@ struct hitsbytarget *hitsbytarget(struct session *se, int *count, NMEM nmem) struct client *cl; size_t sz = 0; + session_enter(se); for (cl = se->clients; cl; cl = client_next_in_session(cl)) sz++; @@ -720,6 +680,7 @@ struct hitsbytarget *hitsbytarget(struct session *se, int *count, NMEM nmem) *count = 0; for (cl = se->clients; cl; cl = client_next_in_session(cl)) { + WRBUF w = wrbuf_alloc(); const char *name = session_setting_oneval(client_get_database(cl), PZ_NAME); @@ -730,19 +691,28 @@ struct hitsbytarget *hitsbytarget(struct session *se, int *count, NMEM nmem) res[*count].diagnostic = client_get_diagnostic(cl); res[*count].state = client_get_state_str(cl); res[*count].connected = client_get_connection(cl) ? 1 : 0; + session_settings_dump(se, client_get_database(cl), w); + res[*count].settings_xml = w; (*count)++; } + session_leave(se); return res; } struct termlist_score **termlist(struct session *s, const char *name, int *num) { int i; + struct termlist_score **tl = 0; + session_enter(s); for (i = 0; i < s->num_termlists; i++) if (!strcmp((const char *) s->termlists[i].name, name)) - return termlist_highscore(s->termlists[i].termlist, num); - return 0; + { + tl = termlist_highscore(s->termlists[i].termlist, num); + break; + } + session_leave(s); + return tl; } #ifdef MISSING_HEADERS @@ -758,13 +728,14 @@ void report_nmem_stats(void) } #endif -struct record_cluster *show_single(struct session *s, const char *id, - struct record_cluster **prev_r, - struct record_cluster **next_r) +struct record_cluster *show_single_start(struct session *s, const char *id, + struct record_cluster **prev_r, + struct record_cluster **next_r) { struct record_cluster *r; - reclist_rewind(s->reclist); + session_enter(s); + reclist_enter(s->reclist); *prev_r = 0; *next_r = 0; while ((r = reclist_read_record(s->reclist))) @@ -772,18 +743,26 @@ struct record_cluster *show_single(struct session *s, const char *id, if (!strcmp(r->recid, id)) { *next_r = reclist_read_record(s->reclist); - return r; + break; } *prev_r = r; } - return 0; + reclist_leave(s->reclist); + if (!r) + session_leave(s); + return r; } -struct record_cluster **show(struct session *s, struct reclist_sortparms *sp, - int start, int *num, int *total, int *sumhits, - NMEM nmem_show) +void show_single_stop(struct session *s, struct record_cluster *rec) { - struct record_cluster **recs = nmem_malloc(nmem_show, *num + session_leave(s); +} + +struct record_cluster **show_range_start(struct session *s, + struct reclist_sortparms *sp, + int start, int *num, int *total, Odr_int *sumhits) +{ + struct record_cluster **recs = nmem_malloc(s->nmem, *num * sizeof(struct record_cluster *)); struct reclist_sortparms *spp; int i; @@ -791,6 +770,7 @@ struct record_cluster **show(struct session *s, struct reclist_sortparms *sp, yaz_timing_t t = yaz_timing_create(); #endif + session_enter(s); if (!s->relevance) { *num = 0; @@ -808,7 +788,8 @@ struct record_cluster **show(struct session *s, struct reclist_sortparms *sp, } reclist_sort(s->reclist, sp); - *total = s->reclist->num_records; + reclist_enter(s->reclist); + *total = reclist_get_num_records(s->reclist); *sumhits = s->total_hits; for (i = 0; i < start; i++) @@ -829,6 +810,7 @@ struct record_cluster **show(struct session *s, struct reclist_sortparms *sp, } recs[i] = r; } + reclist_leave(s->reclist); } #if USE_TIMING yaz_timing_stop(t); @@ -840,6 +822,11 @@ struct record_cluster **show(struct session *s, struct reclist_sortparms *sp, return recs; } +void show_range_stop(struct session *s, struct record_cluster **recs) +{ + session_leave(s); +} + void statistics(struct session *se, struct statistics *stat) { struct client *cl; @@ -867,75 +854,39 @@ void statistics(struct session *se, struct statistics *stat) stat->num_clients = count; } -int start_http_listener(struct conf_config *conf, - const char *listener_override, - const char *proxy_override) +static struct record_metadata *record_metadata_init( + NMEM nmem, const char *value, enum conf_metadata_type type, + struct _xmlAttr *attr) { - struct conf_server *ser; - for (ser = conf->servers; ser; ser = ser->next) + struct record_metadata *rec_md = record_metadata_create(nmem); + struct record_metadata_attr **attrp = &rec_md->attributes; + + for (; attr; attr = attr->next) { - char hp[128]; - *hp = '\0'; - if (listener_override) - { - strcpy(hp, listener_override); - listener_override = 0; /* only first server is overriden */ - } - else - { - strcpy(hp, ser->host ? ser->host : ""); - if (ser->port) - { - if (*hp) - strcat(hp, ":"); - sprintf(hp + strlen(hp), "%d", ser->port); - } - } - if (http_init(hp, ser)) - return -1; - - *hp = '\0'; - if (proxy_override) - strcpy(hp, proxy_override); - else if (ser->proxy_host || ser->proxy_port) + if (attr->children && attr->children->content) { - strcpy(hp, ser->proxy_host ? ser->proxy_host : ""); - if (ser->proxy_port) - { - if (*hp) - strcat(hp, ":"); - sprintf(hp + strlen(hp), "%d", ser->proxy_port); + if (strcmp((const char *) attr->name, "type")) + { /* skip the "type" attribute.. Its value is already part of + the element in output (md-%s) and so repeating it here + is redundant */ + *attrp = nmem_malloc(nmem, sizeof(**attrp)); + (*attrp)->name = + nmem_strdup(nmem, (const char *) attr->name); + (*attrp)->value = + nmem_strdup(nmem, (const char *) attr->children->content); + attrp = &(*attrp)->next; } } - if (*hp) - http_set_proxyaddr(hp, ser->myurl ? ser->myurl : ""); } - return 0; -} + *attrp = 0; -// Master list of connections we're handling events to -static IOCHAN channel_list = 0; -void pazpar2_add_channel(IOCHAN chan) -{ - chan->next = channel_list; - channel_list = chan; -} - -void pazpar2_event_loop() -{ - event_loop(&channel_list); -} - -static struct record_metadata *record_metadata_init( - NMEM nmem, char *value, enum conf_metadata_type type) -{ - struct record_metadata *rec_md = record_metadata_create(nmem); if (type == Metadata_type_generic) { - char * p = value; + char *p = nmem_strdup(nmem, value); + p = normalize7bit_generic(p, " ,/.:(["); - rec_md->data.text.disp = nmem_strdup(nmem, p); + rec_md->data.text.disp = p; rec_md->data.text.sort = 0; } else if (type == Metadata_type_year || type == Metadata_type_date) @@ -956,15 +907,60 @@ static struct record_metadata *record_metadata_init( return rec_md; } +static int get_mergekey_from_doc(xmlDoc *doc, xmlNode *root, const char *name, + struct conf_service *service, WRBUF norm_wr) +{ + xmlNode *n; + int no_found = 0; + for (n = root->children; n; n = n->next) + { + if (n->type != XML_ELEMENT_NODE) + continue; + if (!strcmp((const char *) n->name, "metadata")) + { + xmlChar *type = xmlGetProp(n, (xmlChar *) "type"); + if (!strcmp(name, (const char *) type)) + { + xmlChar *value = xmlNodeListGetString(doc, n->children, 1); + if (value) + { + const char *norm_str; + pp2_relevance_token_t prt = + pp2_relevance_tokenize( + service->mergekey_pct, + (const char *) value, 0); + + if (wrbuf_len(norm_wr) > 0) + wrbuf_puts(norm_wr, " "); + wrbuf_puts(norm_wr, name); + while ((norm_str = + pp2_relevance_token_next(prt))) + { + if (*norm_str) + { + wrbuf_puts(norm_wr, " "); + wrbuf_puts(norm_wr, norm_str); + } + } + xmlFree(value); + pp2_relevance_token_destroy(prt); + no_found++; + } + } + xmlFree(type); + } + } + return no_found; +} + static const char *get_mergekey(xmlDoc *doc, struct client *cl, int record_no, struct conf_service *service, NMEM nmem) { char *mergekey_norm = 0; xmlNode *root = xmlDocGetRootElement(doc); WRBUF norm_wr = wrbuf_alloc(); - xmlNode *n; - /* create mergekey based on mergekey attribute from XSL (if any) */ + /* consider mergekey from XSL first */ xmlChar *mergekey = xmlGetProp(root, (xmlChar *) "mergekey"); if (mergekey) { @@ -972,7 +968,7 @@ static const char *get_mergekey(xmlDoc *doc, struct client *cl, int record_no, pp2_relevance_token_t prt = pp2_relevance_tokenize( service->mergekey_pct, - (const char *) mergekey); + (const char *) mergekey, 0); while ((norm_str = pp2_relevance_token_next(prt))) { @@ -986,53 +982,25 @@ static const char *get_mergekey(xmlDoc *doc, struct client *cl, int record_no, pp2_relevance_token_destroy(prt); xmlFree(mergekey); } - /* append (if any) mergekey=yes metadata values */ - for (n = root->children; n; n = n->next) + else { - if (n->type != XML_ELEMENT_NODE) - continue; - if (!strcmp((const char *) n->name, "metadata")) + /* no mergekey defined in XSL. Look for mergekey metadata instead */ + int field_id; + for (field_id = 0; field_id < service->num_metadata; field_id++) { - struct conf_metadata *ser_md = 0; - int md_field_id = -1; - - xmlChar *type = xmlGetProp(n, (xmlChar *) "type"); - - if (!type) - continue; - - md_field_id - = conf_service_metadata_field_id(service, - (const char *) type); - if (md_field_id >= 0) + struct conf_metadata *ser_md = &service->metadata[field_id]; + if (ser_md->mergekey != Metadata_mergekey_no) { - ser_md = &service->metadata[md_field_id]; - if (ser_md->mergekey == Metadata_mergekey_yes) + int r = get_mergekey_from_doc(doc, root, ser_md->name, + service, norm_wr); + if (r == 0 && ser_md->mergekey == Metadata_mergekey_required) { - xmlChar *value = xmlNodeListGetString(doc, n->children, 1); - if (value) - { - const char *norm_str; - pp2_relevance_token_t prt = - pp2_relevance_tokenize( - service->mergekey_pct, - (const char *) value); - - while ((norm_str = pp2_relevance_token_next(prt))) - { - if (*norm_str) - { - if (wrbuf_len(norm_wr)) - wrbuf_puts(norm_wr, " "); - wrbuf_puts(norm_wr, norm_str); - } - } - xmlFree(value); - pp2_relevance_token_destroy(prt); - } + /* no mergekey on this one and it is required.. + Generate unique key instead */ + wrbuf_rewind(norm_wr); + break; } } - xmlFree(type); } } @@ -1048,57 +1016,139 @@ static const char *get_mergekey(xmlDoc *doc, struct client *cl, int record_no, return mergekey_norm; } +/** \brief see if metadata for pz:recordfilter exists + \param root xml root element of normalized record + \param sdb session database for client + \retval 0 if there is no metadata for pz:recordfilter + \retval 1 if there is metadata for pz:recordfilter + + If there is no pz:recordfilter defined, this function returns 1 + as well. +*/ + +static int check_record_filter(xmlNode *root, struct session_database *sdb) +{ + int match = 0; + xmlNode *n; + const char *s; + s = session_setting_oneval(sdb, PZ_RECORDFILTER); + + if (!s || !*s) + return 1; + + for (n = root->children; n; n = n->next) + { + if (n->type != XML_ELEMENT_NODE) + continue; + if (!strcmp((const char *) n->name, "metadata")) + { + xmlChar *type = xmlGetProp(n, (xmlChar *) "type"); + if (type) + { + size_t len; + const char *eq = strchr(s, '~'); + if (eq) + len = eq - s; + else + len = strlen(s); + if (len == strlen((const char *)type) && + !memcmp((const char *) type, s, len)) + { + xmlChar *value = xmlNodeGetContent(n); + if (value && *value) + { + if (!eq || strstr((const char *) value, eq+1)) + match = 1; + } + xmlFree(value); + } + xmlFree(type); + } + } + } + return match; +} + +static int ingest_to_cluster(struct client *cl, + xmlDoc *xdoc, + xmlNode *root, + int record_no, + const char *mergekey_norm); /** \brief ingest XML record \param cl client holds the result set for record \param rec record buffer (0 terminated) \param record_no record position (1, 2, ..) - \returns resulting record or NULL on failure + \retval 0 OK + \retval -1 failure */ -struct record *ingest_record(struct client *cl, const char *rec, - int record_no) +int ingest_record(struct client *cl, const char *rec, + int record_no, NMEM nmem) { - xmlDoc *xdoc = normalize_record(client_get_database(cl), - client_get_session(cl), rec); - xmlNode *root, *n; - struct record *record; - struct record_cluster *cluster; + struct session_database *sdb = client_get_database(cl); struct session *se = client_get_session(cl); - const char *mergekey_norm; - xmlChar *type = 0; - xmlChar *value = 0; struct conf_service *service = se->service; + xmlDoc *xdoc = normalize_record(sdb, service, rec, nmem); + xmlNode *root; + const char *mergekey_norm; + int ret; if (!xdoc) - return 0; + return -1; root = xmlDocGetRootElement(xdoc); - mergekey_norm = get_mergekey(xdoc, cl, record_no, service, se->nmem); + if (!check_record_filter(root, sdb)) + { + yaz_log(YLOG_WARN, "Filtered out record no %d from %s", record_no, + sdb->database->url); + xmlFreeDoc(xdoc); + return -1; + } + + mergekey_norm = get_mergekey(xdoc, cl, record_no, service, nmem); if (!mergekey_norm) { yaz_log(YLOG_WARN, "Got no mergekey"); xmlFreeDoc(xdoc); - return 0; + return -1; } - record = record_create(se->nmem, - service->num_metadata, service->num_sortkeys, cl, - record_no); - - cluster = reclist_insert(se->reclist, - service, - record, (char *) mergekey_norm, - &se->total_merged); + session_enter(se); + ret = ingest_to_cluster(cl, xdoc, root, record_no, mergekey_norm); + session_leave(se); + + xmlFreeDoc(xdoc); + + return ret; +} + +static int ingest_to_cluster(struct client *cl, + xmlDoc *xdoc, + xmlNode *root, + int record_no, + const char *mergekey_norm) +{ + xmlNode *n; + xmlChar *type = 0; + xmlChar *value = 0; + struct session_database *sdb = client_get_database(cl); + struct session *se = client_get_session(cl); + struct conf_service *service = se->service; + struct record *record = record_create(se->nmem, + service->num_metadata, + service->num_sortkeys, cl, + record_no); + struct record_cluster *cluster = reclist_insert(se->reclist, + service, + record, + mergekey_norm, + &se->total_merged); + if (!cluster) + return -1; if (global_parameters.dump_records) yaz_log(YLOG_LOG, "Cluster id %s from %s (#%d)", cluster->recid, - client_get_database(cl)->database->url, record_no); - if (!cluster) - { - /* no room for record */ - xmlFreeDoc(xdoc); - return 0; - } + sdb->database->url, record_no); relevance_newrec(se->relevance, cluster); // now parsing XML record and adding data to cluster or record metadata @@ -1149,8 +1199,8 @@ struct record *ingest_record(struct client *cl, const char *rec, } // non-merged metadata - rec_md = record_metadata_init(se->nmem, (char *) value, - ser_md->type); + rec_md = record_metadata_init(se->nmem, (const char *) value, + ser_md->type, n->properties); if (!rec_md) { yaz_log(YLOG_WARN, "bad metadata data '%s' for element '%s'", @@ -1163,8 +1213,8 @@ struct record *ingest_record(struct client *cl, const char *rec, *wheretoput = rec_md; // merged metadata - rec_md = record_metadata_init(se->nmem, (char *) value, - ser_md->type); + rec_md = record_metadata_init(se->nmem, (const char *) value, + ser_md->type, 0); wheretoput = &cluster->metadata[md_field_id]; // and polulate with data: @@ -1202,11 +1252,11 @@ struct record *ingest_record(struct client *cl, const char *rec, prt = pp2_relevance_tokenize( service->sort_pct, - rec_md->data.text.disp); + rec_md->data.text.disp, skip_article); pp2_relevance_token_next(prt); - sort_str = pp2_get_sort(prt, skip_article); + sort_str = pp2_get_sort(prt); cluster->sortkeys[sk_field_id]->text.disp = rec_md->data.text.disp; @@ -1257,7 +1307,8 @@ struct record *ingest_record(struct client *cl, const char *rec, // ranking of _all_ fields enabled ... if (ser_md->rank) relevance_countwords(se->relevance, cluster, - (char *) value, ser_md->rank); + (char *) value, ser_md->rank, + ser_md->name); // construct facets ... if (ser_md->termlist) @@ -1295,16 +1346,12 @@ struct record *ingest_record(struct client *cl, const char *rec, if (value) xmlFree(value); - xmlFreeDoc(xdoc); - relevance_donerecord(se->relevance, cluster); se->total_records++; - return record; + return 0; } - - /* * Local variables: * c-basic-offset: 4