-/* $Id: extract.c,v 1.266 2007-10-30 19:17:15 adam Exp $
- Copyright (C) 1995-2007
- Index Data ApS
-
-This file is part of the Zebra server.
+/* This file is part of the Zebra server.
+ Copyright (C) 1995-2008 Index Data
Zebra is free software; you can redistribute it and/or modify it under
the terms of the GNU General Public License as published by the Free
}
}
-static void extract_flush_record_keys(ZebraHandle zh, zint sysno,
- int cmd, zebra_rec_keys_t reckeys,
- zint staticrank);
static void extract_flush_sort_keys(ZebraHandle zh, zint sysno,
int cmd, zebra_rec_keys_t skp);
static void extract_schema_add(struct recExtractCtrl *p, Odr_oid *oid);
static void extract_token_add(RecWord *p);
-static void extract_token_add2(RecWord *p);
static void check_log_limit(ZebraHandle zh)
{
static void init_extractCtrl(ZebraHandle zh, struct recExtractCtrl *ctrl)
{
- int i;
- for (i = 0; i<256; i++)
- {
- zebra_map_t zm = zebra_map_get(zh->reg->zebra_maps, i);
- if (zebra_maps_is_positioned(zm))
- ctrl->seqno[i] = 1;
- else
- ctrl->seqno[i] = 0;
- }
ctrl->flagShowRecords = !zh->m_flag_rw;
}
}
if (!i)
return;
- if (last && start != last)
+ if (last && start != last && zebra_maps_is_index(zm))
zebra_snippets_appendn(h->snippets, p->seqno, 0, ord,
start, last - start);
}
}
if (!map)
break;
- if (start != last)
+ if (start != last && zebra_maps_is_index(zm))
{
zebra_snippets_appendn(h->snippets, p->seqno, 1, ord,
start, last - start);
p->seqno++;
}
}
- if (start != last)
+ if (start != last && zebra_maps_is_index(zm))
zebra_snippets_appendn(h->snippets, p->seqno, 0, ord,
start, last - start);
start = last;
}
+static void snippet_add_icu(RecWord *p, int ord, zebra_map_t zm)
+{
+ struct snip_rec_info *h = p->extractCtrl->handle;
+
+ const char *res_buf = 0;
+ size_t res_len = 0;
+
+ const char *display_buf = 0;
+ size_t display_len = 0;
+
+ zebra_map_tokenize_start(zm, p->term_buf, p->term_len);
+ while (zebra_map_tokenize_next(zm, &res_buf, &res_len,
+ &display_buf, &display_len))
+ {
+ if (zebra_maps_is_index(zm))
+ zebra_snippets_appendn(h->snippets, p->seqno, 0, ord,
+ display_buf, display_len);
+ p->seqno++;
+ }
+}
+
static void snippet_token_add(RecWord *p)
{
struct snip_rec_info *h = p->extractCtrl->handle;
ZebraHandle zh = h->zh;
- zebra_map_t zm = zebra_map_get(zh->reg->zebra_maps, *p->index_type);
+ zebra_map_t zm = zebra_map_get(zh->reg->zebra_maps, p->index_type);
- if (zm && zebra_maps_is_index(zm))
+ if (zm)
{
ZebraExplainInfo zei = zh->reg->zei;
int ch = zebraExplain_lookup_attr_str(
zei, zinfo_index_category_index, p->index_type, p->index_name);
- if (zebra_maps_is_complete(zm))
- snippet_add_complete_field(p, ch, zm);
+ if (zebra_maps_is_icu(zm))
+ snippet_add_icu(p, ch, zm);
else
- snippet_add_incomplete_field(p, ch, zm);
+ {
+ if (zebra_maps_is_complete(zm))
+ snippet_add_complete_field(p, ch, zm);
+ else
+ snippet_add_incomplete_field(p, ch, zm);
+ }
}
}
struct recordGroup *rGroup;
};
-static void all_matches_add(struct recExtractCtrl *ctrl)
+/** \brief add the always-matches index entry and map to real record ID
+ \param ctrl record control
+ \param record_id custom record ID
+ \param sysno system record ID
+
+ This function serves two purposes.. It adds the always matches
+ entry and makes a pointer from the custom record ID (if defined)
+ back to the system record ID (sysno)
+ See zebra_recid_to_sysno .
+ */
+static void all_matches_add(struct recExtractCtrl *ctrl, zint record_id,
+ zint sysno)
{
RecWord word;
extract_init(ctrl, &word);
+ word.record_id = record_id;
+ /* we use the seqno as placeholder for a way to get back to
+ record database from _ALLRECORDS.. This is used if a custom
+ RECORD was defined */
+ word.seqno = sysno;
word.index_name = "_ALLRECORDS";
word.index_type = "w";
- word.seqno = 1;
+
extract_add_index_string(&word, zinfo_index_category_alwaysmatches,
"", 0);
}
ZEBRA_RES zebra_extract_file(ZebraHandle zh, zint *sysno, const char *fname,
- int deleteFlag)
+ enum zebra_recctrl_action_t action)
{
ZEBRA_RES r = ZEBRA_OK;
int i, fd;
default:
yaz_log(YLOG_WARN, "Bad filter version: %s", zh->m_record_type);
}
- if (sysno && deleteFlag)
+ if (sysno && (action == action_delete || action == action_a_delete))
{
streamp = 0;
fi = 0;
zebra_create_stream_fd(streamp, fd, 0);
}
r = zebra_extract_records_stream(zh, streamp,
- deleteFlag ?
- action_delete : action_update,
+ action,
0, /* tst_mode */
zh->m_record_type,
sysno,
stream->endf(stream, &null_offset);;
extractCtrl.init = extract_init;
- if (zh->reg->index_types)
- {
- extractCtrl.tokenAdd = extract_token_add2;
- }
- else
- {
- extractCtrl.tokenAdd = extract_token_add;
- }
+ extractCtrl.tokenAdd = extract_token_add;
extractCtrl.schemaAdd = extract_schema_add;
extractCtrl.dh = zh->reg->dh;
extractCtrl.handle = zh;
else
end_offset = stream->tellf(stream);
- all_matches_add(&extractCtrl);
-
if (extractCtrl.match_criteria[0])
match_criteria = extractCtrl.match_criteria;
}
}
}
}
+
if (zebra_rec_keys_empty(zh->reg->keys))
{
/* the extraction process returned no information - the record
if (! *sysno)
{
- /* new record */
+ /* new record AKA does not exist already */
if (action == action_delete)
{
- yaz_log(YLOG_LOG, "delete %s %s " ZINT_FORMAT, recordType,
- pr_fname, (zint) start_offset);
+ yaz_log(YLOG_LOG, "delete %s %s " ZINT_FORMAT, recordType,
+ pr_fname, (zint) start_offset);
yaz_log(YLOG_WARN, "cannot delete record above (seems new)");
return ZEBRA_FAIL;
}
+ else if (action == action_a_delete)
+ {
+ if (show_progress)
+ yaz_log(YLOG_LOG, "adelete %s %s " ZINT_FORMAT, recordType,
+ pr_fname, (zint) start_offset);
+ return ZEBRA_OK;
+ }
else if (action == action_replace)
{
yaz_log(YLOG_LOG, "update %s %s " ZINT_FORMAT, recordType,
*sysno = rec->sysno;
+
+ if (stream)
+ {
+ all_matches_add(&extractCtrl,
+ zebra_rec_keys_get_custom_record_id(zh->reg->keys),
+ *sysno);
+ }
+
+
recordAttr = rec_init_attr(zh->reg->zei, rec);
if (extractCtrl.staticrank < 0)
{
rec = rec_get(zh->reg->records, *sysno);
assert(rec);
+
+ if (stream)
+ {
+ all_matches_add(&extractCtrl,
+ zebra_rec_keys_get_custom_record_id(zh->reg->keys),
+ *sysno);
+ }
recordAttr = rec_init_attr(zh->reg->zei, rec);
extract_flush_record_keys(zh, *sysno, 0, delkeys,
recordAttr->staticrank);
#endif
- if (action == action_delete)
+ if (action == action_delete || action == action_a_delete)
{
/* record going to be deleted */
#if FLUSH2
zebraExplain_lookup_ord(zh->reg->zei, ord, &index_type,
0/* db */, &string_index);
assert(index_type);
- zebra_term_untrans_iconv(zh, nmem, *index_type,
+ zebra_term_untrans_iconv(zh, nmem, index_type,
&dst_term, str);
*keystr = '\0';
for (i = 0; i<key.len; i++)
yaz_log(log_level_extract, "normal=%d optimized=%d", normal, optimized);
}
-void extract_flush_record_keys(ZebraHandle zh, zint sysno, int cmd,
- zebra_rec_keys_t reckeys,
- zint staticrank)
-{
- ZebraExplainInfo zei = zh->reg->zei;
-
- extract_rec_keys_adjust(zh, cmd, reckeys);
-
- if (log_level_details)
- {
- yaz_log(log_level_details, "Keys for record " ZINT_FORMAT " %s",
- sysno, cmd ? "insert" : "delete");
- extract_rec_keys_log(zh, cmd, reckeys, log_level_details);
- }
- if (!zh->reg->key_block)
- {
- int mem = 1024*1024 * atoi( res_get_def( zh->res, "memmax", "8"));
- const char *key_tmp_dir = res_get_def(zh->res, "keyTmpDir", ".");
- int use_threads = atoi(res_get_def(zh->res, "threads", "1"));
- zh->reg->key_block = key_block_create(mem, key_tmp_dir, use_threads);
- }
- zebraExplain_recordCountIncrement(zei, cmd ? 1 : -1);
-
-#if 0
- yaz_log(YLOG_LOG, "sysno=" ZINT_FORMAT " cmd=%d", sysno, cmd);
- print_rec_keys(zh, reckeys);
-#endif
- if (zebra_rec_keys_rewind(reckeys))
- {
- size_t slen;
- const char *str;
- struct it_key key_in;
- while(zebra_rec_keys_read(reckeys, &str, &slen, &key_in))
- {
- key_block_write(zh->reg->key_block, sysno,
- &key_in, cmd, str, slen,
- staticrank, zh->m_staticrank);
- }
- }
-}
-
-ZEBRA_RES zebra_rec_keys_to_snippets(ZebraHandle zh,
+ZEBRA_RES zebra_rec_keys_to_snippets1(ZebraHandle zh,
zebra_rec_keys_t reckeys,
zebra_snippets *snippets)
{
zebraExplain_lookup_ord(zh->reg->zei, ord, &index_type,
0/* db */, 0 /* string_index */);
assert(index_type);
- zebra_term_untrans_iconv(zh, nmem, *index_type,
+ zebra_term_untrans_iconv(zh, nmem, index_type,
&dst_term, str);
zebra_snippets_append(snippets, seqno, 0, ord, dst_term);
nmem_reset(nmem);
seqno = key.mem[key.len-1];
- zebra_term_untrans(zh, *index_type, dst_buf, str);
+ zebra_term_untrans(zh, index_type, dst_buf, str);
yaz_log(YLOG_LOG, "ord=%d seqno=" ZINT_FORMAT
" term=%s", ord, seqno, dst_buf);
if (!p->index_name)
return;
+ if (log_level_details)
+ {
+ WRBUF w = wrbuf_alloc();
+
+ wrbuf_write_escaped(w, string, length);
+ yaz_log(log_level_details, "extract_add_string: %s", wrbuf_cstr(w));
+ wrbuf_destroy(w);
+ }
if (zebra_maps_is_index(zm))
{
extract_add_index_string(p, zinfo_index_category_index,
extract_add_string(p, zm, buf, i);
}
-static void extract_token_add2_index(ZebraHandle zh, zebra_index_type_t type,
- RecWord *p)
+static void extract_add_icu(RecWord *p, zebra_map_t zm)
{
- struct it_key key;
const char *res_buf = 0;
size_t res_len = 0;
- int r = zebra_index_type_tokenize(type, p->term_buf, p->term_len,
- &res_buf, &res_len);
- int cat = zinfo_index_category_index;
- int ch = zebraExplain_lookup_attr_str(zh->reg->zei, cat, p->index_type, p->index_name);
- if (ch < 0)
- ch = zebraExplain_add_attr_str(zh->reg->zei, cat, p->index_type, p->index_name);
- while (r)
- {
- int i = 0;
- key.mem[i++] = ch;
- key.mem[i++] = p->record_id;
- key.mem[i++] = p->section_id;
-
- if (zh->m_segment_indexing)
- key.mem[i++] = p->segment;
- key.mem[i++] = p->seqno;
- key.len = i;
- yaz_log(YLOG_LOG, "keys_write %.*s", (int) res_len, res_buf);
- zebra_rec_keys_write(zh->reg->keys, res_buf, res_len, &key);
-
+ zebra_map_tokenize_start(zm, p->term_buf, p->term_len);
+ while (zebra_map_tokenize_next(zm, &res_buf, &res_len, 0, 0))
+ {
+ extract_add_string(p, zm, res_buf, res_len);
p->seqno++;
- r = zebra_index_type_tokenize(type, 0, 0, &res_buf, &res_len);
}
}
-static void extract_token_add2(RecWord *p)
-{
- ZebraHandle zh = p->extractCtrl->handle;
- zebra_index_type_t type = zebra_index_type_get(zh->reg->index_types, p->index_type);
- if (type)
- {
- if (zebra_index_type_is_index(type))
- {
- extract_token_add2_index(zh, type, p);
- }
- else if (zebra_index_type_is_sort(type))
- {
- ;
-
- }
- }
-}
/** \brief top-level indexing handler for recctrl system
\param p token data to be indexed
Call sequence:
- extract_token
- zebra_add_{in}_complete
+ extract_token_add
+ extract_add_{in}_complete / extract_add_icu
extract_add_string
extract_add_index_string
static void extract_token_add(RecWord *p)
{
ZebraHandle zh = p->extractCtrl->handle;
- zebra_map_t zm = zebra_map_get_or_add(zh->reg->zebra_maps, *p->index_type);
+ zebra_map_t zm = zebra_map_get_or_add(zh->reg->zebra_maps, p->index_type);
WRBUF wrbuf;
if (log_level_details)
}
if ((wrbuf = zebra_replace(zm, 0, p->term_buf, p->term_len)))
{
- p->term_buf = wrbuf_buf(wrbuf);
- p->term_len = wrbuf_len(wrbuf);
+ p->term_buf = wrbuf_buf(wrbuf);
+ p->term_len = wrbuf_len(wrbuf);
+ }
+ if (zebra_maps_is_icu(zm))
+ {
+ extract_add_icu(p, zm);
}
- if (zebra_maps_is_complete(zm))
- extract_add_complete_field(p, zm);
else
- extract_add_incomplete_field(p, zm);
+ {
+ if (zebra_maps_is_complete(zm))
+ extract_add_complete_field(p, zm);
+ else
+ extract_add_incomplete_field(p, zm);
+ }
}
static void extract_set_store_data_cb(struct recExtractCtrl *p,
const char *str;
struct it_key key_in;
- zebra_sort_sysno(si, sysno);
+ NMEM nmem = nmem_create();
+ struct sort_add_ent {
+ int ord;
+ int cmd;
+ struct sort_add_ent *next;
+ WRBUF wrbuf;
+ zint sysno;
+ };
+ struct sort_add_ent *sort_ent_list = 0;
while (zebra_rec_keys_read(reckeys, &str, &slen, &key_in))
{
int ord = CAST_ZINT_TO_INT(key_in.mem[0]);
+ zint filter_sysno = key_in.mem[1];
+
+ struct sort_add_ent **e = &sort_ent_list;
+ while (*e && (*e)->ord != ord)
+ e = &(*e)->next;
+ if (!*e)
+ {
+ *e = nmem_malloc(nmem, sizeof(**e));
+ (*e)->next = 0;
+ (*e)->wrbuf = wrbuf_alloc();
+ (*e)->ord = ord;
+ (*e)->cmd = cmd;
+ (*e)->sysno = filter_sysno ? filter_sysno : sysno;
+ }
- zebra_sort_type(si, ord);
- if (cmd == 1)
- zebra_sort_add(si, str, slen);
- else
- zebra_sort_delete(si);
+ wrbuf_write((*e)->wrbuf, str, slen);
+ wrbuf_putc((*e)->wrbuf, '\0');
}
+ if (sort_ent_list)
+ {
+ zint last_sysno = 0;
+ struct sort_add_ent *e = sort_ent_list;
+ for (; e; e = e->next)
+ {
+ if (last_sysno != e->sysno)
+ {
+ zebra_sort_sysno(si, e->sysno);
+ last_sysno = e->sysno;
+ }
+ zebra_sort_type(si, e->ord);
+ if (e->cmd == 1)
+ zebra_sort_add(si, e->wrbuf);
+ else
+ zebra_sort_delete(si);
+ wrbuf_destroy(e->wrbuf);
+ }
+ }
+ nmem_destroy(nmem);
}
}