blob: e72cf4c61d770594e1011e11bcdc006559bcf234 [file] [log] [blame]
#include <collocatordb.h>
#include <math.h>
#include <pthread.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <sys/mman.h>
#include <fcntl.h>
#include <unistd.h>
#include <perl.h>
#define max_size 2000
#define max_w 50
#define MAX_NEIGHBOURS 1000
#define MAX_TARGET_WORDS 100
#define MAX_WORDS -1
#define MAX_THREADS 100
#define MAX_CC 50
#define EXP_TABLE_SIZE 1000
#define MAX_EXP 6
#define MIN_RESP 0.50
/* Per request diagnostics. Every neighbourhood request used to write a few
hundred lines - one per window position, one per vocabulary lookup, the
whole JSON of a similar profile - and a busy instance turned that into a
million lines a day, which is what filled the log partition of the
production machine. They are off unless DEREKOVECS_DEBUG is set to
something other than 0 or the empty string. Startup messages and error
messages are not affected. */
static int derekovecs_debug(void) {
static int enabled = -1;
if (enabled < 0) {
const char *e = getenv("DEREKOVECS_DEBUG");
enabled = (e && *e && strcmp(e, "0") != 0) ? 1 : 0;
}
return enabled;
}
#define DEBUG_PRINTF(...) do { if (derekovecs_debug()) printf(__VA_ARGS__); } while (0)
#define DEBUG_EPRINTF(...) do { if (derekovecs_debug()) fprintf(stderr, __VA_ARGS__); } while (0)
#define DEBUG_FFLUSH() do { if (derekovecs_debug()) fflush(stdout); } while (0)
//the thread function
void *connection_handler(void *);
typedef struct {
long long wordi;
long position;
float activation;
float average;
float cprobability; // column wise probability
float cprobability_sum;
float probability;
float activation_sum;
float max_activation;
/* The score before the sigmoid, q . syn1neg[collocate, position]. activation
is expTable[z], quantized in steps of 1/83 and capped at MAX_EXP; z is
not. raw is the value at one position, max_raw the maximum over
positions. */
float raw;
float max_raw;
float heat[16];
} collocator;
typedef struct {
collocator *best;
int length;
} knn;
typedef struct {
long long wordi[MAX_TARGET_WORDS];
/* '+' or '-': the sign with which wordi[i] enters the query vector */
char sep[MAX_TARGET_WORDS];
/* blank separated tokens of the query that are not in the vocabulary */
char oov[max_size];
int length; /* number of tokens found in the vocabulary */
int subtractions; /* how many of them enter with a '-' */
} wordlist;
typedef struct {
long cutoff;
wordlist *wl;
char *token;
int N;
long from;
unsigned long upto;
collocator *best;
float *target_sums;
float *window_sums;
float threshold;
} knnpars;
typedef struct {
uint32_t index;
float value;
} sparse_t;
typedef struct {
uint32_t len;
sparse_t nbr[100];
} profile_t;
float *M, *M2 = 0L, *syn1neg_window, *expTable;
char *vocab;
char *garbage = NULL;
COLLOCATORDB *cdb = NULL;
profile_t *sprofiles = NULL;
size_t sprofiles_qty = 0;
long long words, size, merged_end;
long long merge_words = 0;
int num_threads = 20;
int latin_enc = 0;
int window;
/* load collocation profiles if file exists */
int load_sprofiles(char *vecsname) {
char *basename = strdup(vecsname);
char *pos = strstr(basename, ".vecs");
if (pos)
*pos = 0;
char binsprofiles_fname[256];
strcpy(binsprofiles_fname, basename);
strcat(binsprofiles_fname, ".sprofiles.bin");
FILE *fp = fopen(binsprofiles_fname, "rb");
if (fp == NULL) {
printf("Collocation profiles %s not found. No problem.\n", binsprofiles_fname);
return 0;
}
fseek(fp, 0L, SEEK_END);
size_t sz = ftell(fp);
fclose(fp);
int fd = open(binsprofiles_fname, O_RDONLY);
sprofiles = mmap(0, sz, PROT_READ, MAP_SHARED, fd, 0);
if (sprofiles == MAP_FAILED) {
close(fd);
fprintf(stderr, "Cannot mmap %s\n", binsprofiles_fname);
sprofiles = NULL;
return 0;
} else {
sprofiles_qty = sz / sizeof(profile_t);
fprintf(stderr, "Successfully mmaped %s containing similar profiles for %ld word forms.\n", binsprofiles_fname, sprofiles_qty);
}
return 1;
}
char *removeExtension(char* myStr) {
char *retStr;
char *lastExt;
if (myStr == NULL) return NULL;
if ((retStr = malloc (strlen (myStr) + 1)) == NULL) return NULL;
strcpy (retStr, myStr);
lastExt = strrchr (retStr, '.');
if (lastExt != NULL)
*lastExt = '\0';
return retStr;
}
int init_net(char *file_name, char *net_name, int latin, int do_open_cdb) {
FILE *f, *binvecs, *binwords;
int binwords_fd, binvecs_fd, net_fd, i;
long long a, b;
float len;
double val;
char binvecs_fname[1024], binwords_fname[1024];
if (strstr(file_name, ".txt")) {
strcpy(binwords_fname, removeExtension(file_name));
} else {
strcpy(binwords_fname, file_name);
}
strcat(binwords_fname, ".words");
strcpy(binvecs_fname, file_name);
strcat(binvecs_fname, ".vecs");
latin_enc = latin;
f = fopen(file_name, "rb");
if (f == NULL) {
printf("Input file %s not found\n", file_name);
return -1;
}
fscanf(f, "%lld", &words);
if (MAX_WORDS > 0 && words > MAX_WORDS) words = MAX_WORDS;
fscanf(f, "%lld", &size);
if ((binvecs_fd = open(binvecs_fname, O_RDONLY)) < 0 || (binwords_fd = open(binwords_fname, O_RDONLY)) < 0) {
printf("Converting %s to memory mappable structures\n", file_name);
vocab = (char *)malloc((long long)words * max_w * sizeof(char));
M = (float *)malloc((long long)words * (long long)size * sizeof(float));
if (M == NULL) {
printf("Cannot allocate memory: %lld MB %lld %lld\n", (long long)words * size * sizeof(float) / 1048576, words, size);
return -1;
}
if (strstr(file_name, ".txt")) {
printf("%lld words in ascii vector file with vector size %lld\n", words, size);
for (b = 0; b < words; b++) {
a = 0;
while (1) {
vocab[b * max_w + a] = fgetc(f);
if (feof(f) || (vocab[b * max_w + a] == ' ')) break;
if ((a < max_w) && (vocab[b * max_w + a] != '\n')) a++;
}
vocab[b * max_w + a] = 0;
len = 0;
for (a = 0; a < size; a++) {
fscanf(f, "%lf", &val);
M[a + b * size] = val;
len += val * val;
}
len = sqrt(len);
for (a = 0; a < size; a++) M[a + b * size] /= len;
}
} else {
for (b = 0; b < words; b++) {
a = 0;
while (1) {
vocab[b * max_w + a] = fgetc(f);
if (feof(f) || (vocab[b * max_w + a] == ' ')) break;
if ((a < max_w) && (vocab[b * max_w + a] != '\n')) a++;
}
vocab[b * max_w + a] = 0;
fread(&M[b * size], sizeof(float), size, f);
len = 0;
for (a = 0; a < size; a++) len += M[a + b * size] * M[a + b * size];
len = sqrt(len);
for (a = 0; a < size; a++) M[a + b * size] /= len;
}
}
if ((binvecs = fopen(binvecs_fname, "wb")) != NULL && (binwords = fopen(binwords_fname, "wb")) != NULL) {
fwrite(M, sizeof(float), (long long)words * (long long)size, binvecs);
fclose(binvecs);
fwrite(vocab, sizeof(char), (long long)words * max_w, binwords);
fclose(binwords);
}
}
if ((binvecs_fd = open(binvecs_fname, O_RDONLY)) >= 0 && (binwords_fd = open(binwords_fname, O_RDONLY)) >= 0) {
M = mmap(0, sizeof(float) * (long long)words * (long long)size, PROT_READ, MAP_SHARED, binvecs_fd, 0);
vocab = mmap(0, sizeof(char) * (long long)words * max_w, PROT_READ, MAP_SHARED, binwords_fd, 0);
if (M == MAP_FAILED || vocab == MAP_FAILED) {
close(binvecs_fd);
close(binwords_fd);
fprintf(stderr, "Cannot mmap %s or %s\n", binwords_fname, binvecs_fname);
exit(-1);
}
} else {
fprintf(stderr, "Cannot open %s or %s\n", binwords_fname, binvecs_fname);
exit(-1);
}
fclose(f);
if (net_name && strlen(net_name) > 0) {
if ((net_fd = open(net_name, O_RDONLY)) >= 0) {
window = (lseek(net_fd, 0, SEEK_END) - sizeof(float) * words * size) / words / size / sizeof(float) / 2;
// lseek(net_fd, sizeof(float) * words * size, SEEK_SET);
// munmap(M, sizeof(float) * words * size);
M2 = mmap(0, sizeof(float) * words * size + sizeof(float) * 2 * window * size * words, PROT_READ, MAP_SHARED, net_fd, 0);
if (M2 == MAP_FAILED) {
close(net_fd);
fprintf(stderr, "Cannot mmap %s\n", net_name);
exit(-1);
}
syn1neg_window = M2 + words * size;
} else {
fprintf(stderr, "Cannot open %s\n", net_name);
exit(-1);
}
fprintf(stderr, "Successfully memmaped %s. Determined window size: %d\n", net_name, window);
if (do_open_cdb) {
char collocatordb_name[2048];
strcpy(collocatordb_name, net_name);
char *ext = rindex(collocatordb_name, '.');
if (ext) {
strcpy(ext, ".rocksdb");
if (access(collocatordb_name, R_OK) == 0) {
*ext = 0;
fprintf(stderr, "Opening collocator DB %s\n", collocatordb_name);
cdb = open_collocatordb(collocatordb_name);
} else {
fprintf(stderr, "Cannot open collocator DB %s\n", collocatordb_name);
}
}
}
}
expTable = (float *)malloc((EXP_TABLE_SIZE + 1) * sizeof(float));
for (i = 0; i < EXP_TABLE_SIZE; i++) {
expTable[i] = exp((i / (float)EXP_TABLE_SIZE * 2 - 1) * MAX_EXP); // Precompute the exp() table
expTable[i] = expTable[i] / (expTable[i] + 1); // Precompute f(x) = x / (x + 1)
}
return 0;
}
/* The collocator threads of one request accumulate their activation sums per
window position here. This must not be shared between requests, so every
request allocates its own array. Only valid once init_net() has determined
the window size, i.e. if M2 is set. */
float *new_window_sums() {
return calloc((window + 1) * 2, sizeof(float));
}
long mergeVectors(char *file_name) {
FILE *f;
int binwords_fd, binvecs_fd;
float *merge_vecs;
char *merge_vocab;
/* long long merge_words, merge_size; */
long long merge_size;
char binvecs_fname[256], binwords_fname[256];
strcpy(binwords_fname, file_name);
strcat(binwords_fname, ".words");
strcpy(binvecs_fname, file_name);
strcat(binvecs_fname, ".vecs");
f = fopen(file_name, "rb");
if (f == NULL) {
printf("Input file %s not found\n", file_name);
exit(-1);
}
fscanf(f, "%lld", &merge_words);
fscanf(f, "%lld", &merge_size);
if (merge_size != size) {
fprintf(stderr, "vectors must have the same length\n");
exit(-1);
}
if ((binvecs_fd = open(binvecs_fname, O_RDONLY)) >= 0 && (binwords_fd = open(binwords_fname, O_RDONLY)) >= 0) {
merge_vecs = malloc(sizeof(float) * (words + merge_words) * size);
merge_vocab = malloc(sizeof(char) * (words + merge_words) * max_w);
if (merge_vecs == NULL || merge_vocab == NULL) {
close(binvecs_fd);
close(binwords_fd);
fprintf(stderr, "Cannot reserve memory for %s or %s\n", binwords_fname, binvecs_fname);
exit(-1);
}
read(binvecs_fd, merge_vecs, merge_words * size * sizeof(float));
read(binwords_fd, merge_vocab, merge_words * max_w);
} else {
fprintf(stderr, "Cannot open %s or %s\n", binwords_fname, binvecs_fname);
exit(-1);
}
printf("Successfully reallocated memory\nMerging...\n");
fflush(stdout);
memcpy(merge_vecs + merge_words * size, M, words * size * sizeof(float));
memcpy(merge_vocab + merge_words * max_w, vocab, words * max_w);
munmap(M, words * size * sizeof(float));
munmap(vocab, words * max_w);
M = merge_vecs;
vocab = merge_vocab;
merged_end = merge_words;
words += merge_words;
fclose(f);
printf("merged_end: %lld, words: %lld\n", merged_end, words);
//printBiggestMergedDifferences();
return ((long)merged_end);
}
void filter_garbage() {
long i;
unsigned char *w, previous, c;
garbage = malloc(words);
memset(garbage, 0, words);
for (i = 0; i < words; i++) {
w = (unsigned char *) vocab + i * max_w;
previous = 0;
if (strncmp("quot", (const char *)w, 4) == 0) {
garbage[i] = 1;
// printf("Gargabe: %s\n", vocab + i * max_w);
} else {
while ((c = *w++) && !garbage[i]) {
if (((c <= 90 && c >= 65) && (previous >= 97 && previous <= 122)) ||
(previous == '-' && (c & 32)) ||
(previous == 0xc2 && (c == 0xa4 || c == 0xb6)) ||
(previous == 'q' && c == 'u' && *(w) == 'o' && *(w + 1) == 't') || /* quot */
c == '<') {
garbage[i] = 1;
continue;
}
previous = c;
}
}
}
return;
}
knn *simpleGetCollocators(int word, int number, long cutoff, int *result) {
knnpars *pars = calloc(sizeof(knnpars), 1);
float *target_sums = NULL;
float *my_window_sums = malloc(sizeof(float) * (window + 1) * 2);
pars->cutoff = (cutoff ? cutoff : 300000);
long a;
for (a = 0; a < cutoff; a++)
target_sums[a] = 0;
pars->target_sums = target_sums;
pars->window_sums = my_window_sums;
pars->N = (number ? number : 20);
pars->from = 0;
pars->upto = window * 2 - 1;
knn *syn_nbs = NULL; // = (knn*) getCollocators(pars);
free(pars);
free(my_window_sums);
free(target_sums);
return syn_nbs;
}
/* Frees a knn result as returned by getCollocators() via pthread_exit(). */
void free_knn(knn *nbs) {
if (nbs == NULL)
return;
free(nbs->best);
free(nbs);
}
/* Predicts the collocates of a query. The score is
sigmoid(q . syn1neg[target, position]), linear in q before the sigmoid, so q
may be the signed combination of input vectors that the paradigmatic side
searches around rather than the vector of a single word.
q is the mean over the terms, not their sum: the sigmoid is informative over
a narrow range only, and a sum would push the strongest collocates past
MAX_EXP into saturation. The mean keeps every query in the range MIN_RESP
and the auto focus are calibrated for. One word, and a balanced analogy
(+1 -1 +1), are one term and keep the previous scale. */
void *getCollocators(void *args) {
knnpars *pars = args;
int N = pars->N;
wordlist *wl = pars->wl;
knn *nbs = NULL;
long window_layer_size = size * window * 2;
long a, b, c, d, op, window_offset, target, max_target = 0, maxmax_target;
float f, raw, max_f, maxmax_f, scale;
float qvec[max_size];
int terms = 0;
float *target_sums = NULL, worstbest, wpos_sum;
collocator *best;
if (M2 == NULL || wl == NULL || wl->length < 1)
return NULL;
for (a = 0; a < wl->length; a++) terms += (wl->sep[a] == '-' ? -1 : 1);
scale = (terms > 1 ? 1.0f / terms : 1.0f);
for (c = 0; c < size; c++) qvec[c] = 0;
for (a = 0; a < wl->length; a++) {
long long off = wl->wordi[a] * size;
if (wl->sep[a] == '-')
for (c = 0; c < size; c++) qvec[c] -= M2[off + c];
else
for (c = 0; c < size; c++) qvec[c] += M2[off + c];
}
for (c = 0; c < size; c++) qvec[c] *= scale;
a = posix_memalign((void **)&target_sums, 128, pars->cutoff * sizeof(float));
memset(target_sums, 0, pars->cutoff * sizeof(float));
best = malloc((N > 200 ? N : 200) * sizeof(collocator));
memset(best, 0, (N > 200 ? N : 200) * sizeof(collocator));
worstbest = pars->threshold;
for (b = 0; b < pars->cutoff; b++)
target_sums[b] = 0;
for (b = 0; b < N; b++) {
best[b].wordi = -1;
best[b].probability = 1;
best[b].activation = worstbest;
best[b].raw = 0;
}
d = wl->wordi[0];
maxmax_f = -1;
maxmax_target = 0;
for (a = pars->from; a < pars->upto; a++) {
if (a >= window)
a++;
wpos_sum = 0;
DEBUG_PRINTF("window pos: %ld\n", a);
if (a != window) {
max_f = -1;
window_offset = a * size;
if (a > window)
window_offset -= size;
for (target = 0; target < pars->cutoff; target++) {
if (garbage && garbage[target]) continue;
/* an operand of the query is not a collocate of itself */
for (op = 0; op < wl->length && wl->wordi[op] != target; op++)
;
if (op < wl->length)
continue;
f = 0;
for (c = 0; c < size; c++)
f += qvec[c] * syn1neg_window[target * window_layer_size + window_offset + c];
raw = f;
/* Saturate rather than drop: skipping the tails removed the strongest
collocates from the list and from wpos_sum and target_sums. */
if (f < -MAX_EXP)
f = -MAX_EXP;
else if (f > MAX_EXP)
f = MAX_EXP;
f = expTable[(int)((f + MAX_EXP) * (EXP_TABLE_SIZE / MAX_EXP / 2))];
wpos_sum += f;
target_sums[target] += f;
if (f > worstbest) {
for (b = 0; b < N; b++) {
if (f > best[b].activation) {
memmove(best + b + 1, best + b, (N - b - 1) * sizeof(collocator));
best[b].activation = f;
best[b].raw = raw;
best[b].wordi = target;
best[b].position = window - a;
break;
}
}
if (b == N - 1)
worstbest = best[N - 1].activation;
}
}
DEBUG_PRINTF("%ld %.2f\n", max_target, max_f);
DEBUG_PRINTF("%s (%.2f) ", &vocab[max_target * max_w], max_f);
if (max_f > maxmax_f) {
maxmax_f = max_f;
maxmax_target = max_target;
}
for (b = 0; b < N; b++)
if (best[b].position == window - a)
best[b].cprobability = best[b].activation / wpos_sum;
} else {
DEBUG_PRINTF("\x1b[1m%s\x1b[0m ", &vocab[d * max_w]);
}
pars->window_sums[a] = wpos_sum;
}
for (b = 0; b < pars->cutoff; b++)
pars->target_sums[b] += target_sums[b]; //(target_sums[b] / wpos_sum ) / (window * 2);
free(target_sums);
for (b = 0; b < N && best[b].wordi >= 0; b++)
;
// THIS LOOP IS NEEDED (b...)
// printf("%d: best syn: %s %.2f %.5f\n", b, &vocab[best[b].wordi*max_w], best[b].activation, best[b].probability);
// printf("\n");
nbs = malloc(sizeof(knn));
nbs->best = best;
nbs->length = b - 1;
pthread_exit(nbs);
}
float getOutputWeight(int hidden, long target, int window_position) {
const long window_layer_size = size * window * 2;
int a;
if (window_position == 0 || window_position > window || window_position < -window) {
fprintf(stderr, "window_position: %d - assert: -%d <= window_position <= %d && window_position != 0 failed.\n", window_position, window, window);
exit(-1);
}
if (hidden >= size) {
fprintf(stderr, "hidden: %d - assert: hidden < %lld failed.\n", hidden, size);
exit(-1);
}
if (target >= words) {
fprintf(stderr, "target: %ld - assert: target < %lld failed.\n", target, words);
exit(-1);
}
a = window_position + window;
if (a > window) {
--a;
}
long window_offset = a * size;
return syn1neg_window[target * window_layer_size + window_offset + hidden];
}
/* Returns an SV* (not an AV*) on purpose: for an AV* return value Inline::C
generates newRV(), which leaves the array itself with a reference count of
one after the mortal reference is gone, i.e. it leaks the whole array on
every call. */
SV *getVecs(AV *array) {
int i, b;
AV *result = newAV();
for (i = 0; i <= av_len(array); i++) {
SV **elem = av_fetch(array, i, 0);
if (elem != NULL) {
long j = (long)SvNV(*elem);
AV *vector = newAV();
/* ranks come from the request, reading outside the model would crash */
if (j >= 0 && j < words) {
for (b = 0; b < size; b++) {
av_push(vector, newSVnv(M[b + j * size]));
}
}
av_push(result, newRV_noinc((SV *)vector));
}
}
return newRV_noinc((SV *)result);
}
/* Word ids of the collocator db refer to the primary model, which occupies
[0, words - merged_end) of its own vocabulary. libcollocatordb crashes on
ids outside that range, and the ids come straight from the request. */
int valid_cdb_node(long node) {
return cdb != NULL && node >= 0 && node < words - merged_end;
}
/* All functions handing a string back to perl return an SV*, because for a
char* return value Inline::C only copies the string into the return SV and
never frees the buffer we allocated here. */
SV *getSimilarProfiles(long node) {
int i;
char buffer[120000];
char pair_buffer[2048];
buffer[0] = '[';
buffer[1] = 0;
if (node < 0 || node >= sprofiles_qty) {
DEBUG_PRINTF("Not available in precomputed profile\n");
return newSVpv("[{\"w\":\"not available\", \"v\":0}]\n", 0);
}
DEBUG_PRINTF("******* %s ******\n", &vocab[max_w * node]);
for (i = 0; i < 100 && i < sprofiles[node].len; i++) {
sprintf(pair_buffer, "{\"w\":\"%s\", \"v\":%f},", &vocab[max_w * (sprofiles[node].nbr[i].index)], sprofiles[node].nbr[i].value);
strcat(buffer, pair_buffer);
}
if (i > 0)
buffer[strlen(buffer) - 1] = ']';
else
strcat(buffer, "]");
strcat(buffer, "\n");
DEBUG_PRINTF("%s", buffer);
return newSVpv(buffer, 0);
}
/* get_collocat*_as_json() hand out strdup()ed buffers that we own. */
SV *getCollocationScores(long node, long collocate) {
char *json = NULL;
SV *res;
if (valid_cdb_node(node) && valid_cdb_node(collocate))
json = (char *)get_collocation_scores_as_json(cdb, node, collocate);
res = newSVpv(json ? json : "{\"collocates\":[]}", 0);
free(json);
return res;
}
SV *getClassicCollocators(long node) {
char *json = NULL;
SV *res;
if (valid_cdb_node(node))
json = (char *)get_collocators_as_json(cdb, node);
res = newSVpv(json ? json : "{\"collocates\":[]}", 0);
free(json);
return res;
}
/* Returns the position of a word in the vocabulary, or -1. In a merged model
the two vocabularies sit next to each other and search_backw selects which
of them is looked at. */
static long long lookupWord(const char *word, int search_backw) {
long long b, lower = (merge_words ? merge_words : 0);
long long upper = (merge_words ? merge_words : words);
if (search_backw) {
for (b = words - 1; b >= lower && strcmp(&vocab[b * max_w], word) != 0; b--)
;
if (b < lower) b = -1;
} else {
for (b = 0; b < upper && strcmp(&vocab[b * max_w], word) != 0; b++)
;
if (b >= upper) b = -1;
}
return b;
}
/* Splits a query into the operands of a vector expression. Blanks separate
words as before, and a leading '+' or '-' decides with which sign a word
enters the query vector, so that "König - Mann + Frau" is answered with the
neighbours of vec(König) - vec(Mann) + vec(Frau) rather than with those of
any single word. A sign is only an operator at the beginning of a token,
which leaves hyphenated words such as "Nord-Süd-Dialog" searchable; a
free standing sign applies to the word that follows it.
Tokens that are not in the vocabulary are left out of the list and
collected in wl->oov, so that the caller can report them. */
wordlist *getTargetWords(char *st1, int search_backw) {
wordlist *wl = calloc(1, sizeof(wordlist));
char *copy = strdup(st1), *tok, *saveptr = NULL;
size_t oov_len = 0;
int sign = '+';
if (wl == NULL || copy == NULL) {
free(wl);
free(copy);
return NULL;
}
for (tok = strtok_r(copy, " \t\r\n", &saveptr); tok != NULL; tok = strtok_r(NULL, " \t\r\n", &saveptr)) {
long long b;
if (*tok == '+' || *tok == '-') {
sign = *tok++;
if (*tok == 0) continue; /* " - Mann": the sign belongs to the next token */
}
if (wl->length >= MAX_TARGET_WORDS) break;
/* Counted before the lookup: an unknown operand does not change what was
asked for. */
if (sign == '-') wl->subtractions++;
b = lookupWord(tok, search_backw);
if (b < 0) {
DEBUG_EPRINTF("Out of dictionary word: \"%s\"\n", tok);
if (oov_len + strlen(tok) + 2 <= sizeof(wl->oov)) {
if (oov_len > 0) wl->oov[oov_len++] = ' ';
strcpy(wl->oov + oov_len, tok);
oov_len += strlen(tok);
}
} else {
DEBUG_EPRINTF("Word: \"%s\" Sign: %c Position in vocabulary: %lld\n", &vocab[b * max_w], sign, b);
wl->sep[wl->length] = (char)sign;
wl->wordi[wl->length++] = b;
}
sign = '+';
}
free(copy);
return (wl);
}
long getWordNumber(char *word) {
wordlist *wl = getTargetWords(word, 0);
long res = 0;
if (wl == NULL)
return(0);
if(wl->length > 0)
res = wl->wordi[0];
free(wl);
return(res);
}
float get_distance(long b, long c) {
long a;
float dist = 0;
for (a = 0; a < size; a++) dist += M[a + c * size] * M[a + b * size];
return dist;
}
/* The result is computed once and then kept in a static buffer for the
lifetime of the process. */
SV *getBiggestMergedDifferences() {
static char *result = NULL;
float dist;
long long a, c;
int N = 1000;
if (merged_end == 0)
result = "[]";
if (result != NULL)
return newSVpv(result, 0);
DEBUG_PRINTF("Looking for biggest distances between main and merged vectors ...\n");
collocator *best;
best = malloc(N * sizeof(collocator));
memset(best, 0, N * sizeof(collocator));
float worstbest = 1000000;
for (a = 0; a < N; a++) best[a].activation = worstbest;
for (c = 0; c < 500000; c++) {
if (garbage && garbage[c]) continue;
dist = 0;
for (a = 0; a < size; a++) dist += M[a + c * size] * M[a + (c + merged_end) * size];
if (dist < worstbest) {
for (a = 0; a < N; a++) {
if (dist < best[a].activation) {
memmove(best + a + 1, best + a, (N - a - 1) * sizeof(collocator));
best[a].activation = dist;
best[a].wordi = c;
break;
}
}
worstbest = best[N - 1].activation;
}
}
result = malloc(N * (max_w + 64));
char *p = (char *) result;
*p++ = '[';
*p = 0;
for (a = 0; a < N; a++) {
p += sprintf(p, "{\"rank\":%lld,\"word\":\"%s\",\"dist\":%.3f},", a, &vocab[best[a].wordi * max_w], 1 - best[a].activation);
}
*--p = ']';
free(best);
return newSVpv(result, 0);
}
float cos_similarity(long b, long c) {
float dist = 0;
long a;
for (a = 0; a < size; a++) dist += M[b * size + a] * M[c * size + a];
return dist;
}
SV *cos_similarity_as_json(char *w1, char *w2) {
wordlist *a, *b;
float res;
char json[32];
a = getTargetWords(w1, 0);
b = getTargetWords(w2, 0);
if (a == NULL || b == NULL || a->length != 1 || b->length != 1)
res = -1;
else {
res = cos_similarity(a->wordi[0], b->wordi[0]);
DEBUG_EPRINTF("a: %lld b: %lld res:%f\n", a->wordi[0], b->wordi[0], res);
}
free(a);
free(b);
sprintf(json, "%.5f", res);
return newSVpv(json, 0);
}
void *_get_neighbours(void *arg) {
knnpars *pars = arg;
int N = pars->N;
long from = pars->from;
unsigned long upto = pars->upto;
char *sep;
float dist, len, vec[max_size];
long long a, b, c, cn, *bi;
knn *nbs = NULL;
wordlist *wl = pars->wl;
collocator *best = pars->best;
float worstbest = -1;
for (a = 0; a < N; a++) {
best[a].activation = -1;
best[a].wordi = -1;
}
bi = wl->wordi;
cn = wl->length;
sep = wl->sep;
if (cn < 1) {
goto end;
}
for (a = 0; a < size; a++) vec[a] = 0;
for (b = 0; b < cn; b++) {
if (sep[b] == '-')
for (a = 0; a < size; a++) vec[a] -= M[a + bi[b] * size];
else
for (a = 0; a < size; a++) vec[a] += M[a + bi[b] * size];
}
len = 0;
for (a = 0; a < size; a++) len += vec[a] * vec[a];
len = sqrt(len);
/* An expression whose operands cancel each other out, "Haus - Haus", has no
position to search around. */
if (len == 0) {
goto end;
}
for (a = 0; a < size; a++) vec[a] /= len;
for (c = from; c < upto; c++) {
if (garbage && garbage[c]) continue;
a = 0;
// do not skip taget word
// for (b = 0; b < cn; b++) if (bi[b] == c) a = 1;
// if (a == 1) continue;
dist = 0;
for (a = 0; a < size; a++) dist += vec[a] * M[a + c * size];
if (dist > worstbest) {
for (a = 0; a < N; a++) {
if (dist > best[a].activation) {
memmove(best + a + 1, best + a, (N - a - 1) * sizeof(collocator));
best[a].activation = dist;
best[a].wordi = c;
break;
}
}
worstbest = best[N - 1].activation;
}
}
end:
pthread_exit(nbs);
}
int cmp_activation(const void *a, const void *b) {
float fb = ((collocator *)a)->activation;
float fa = ((collocator *)b)->activation;
return (fa > fb) - (fa < fb);
}
int cmp_probability(const void *a, const void *b) {
float fb = ((collocator *)a)->probability;
float fa = ((collocator *)b)->probability;
return (fa > fb) - (fa < fb);
}
SV *getPosWiseW2VCollocators(char *word, long maxPerPos, long cutoff, float threshold, const char *format) {
float *target_sums = NULL;
float *window_sums = NULL;
long a, b, entries = 0;
knn *syn_nbs[MAX_THREADS];
knnpars pars[MAX_THREADS];
pthread_t *pt = NULL;
wordlist *wl = NULL;
int syn_threads = (M2 ? window * 2 : 0);
int search_backw = 0;
char *result = NULL;
SV *res_sv;
for (a = 0; a < MAX_THREADS; a++) syn_nbs[a] = NULL;
if (cutoff < 1 || cutoff > words)
cutoff = words;
wl = getTargetWords(word, search_backw);
if (wl == NULL || wl->length < 1 || syn_threads < 1) {
free(wl);
return newSVpv("", 0);
}
pt = (pthread_t *)malloc((num_threads + 1) * sizeof(pthread_t));
window_sums = new_window_sums();
a = posix_memalign((void **)&target_sums, 128, cutoff * sizeof(float));
memset(target_sums, 0, cutoff * sizeof(float));
DEBUG_PRINTF("Starting %d threads\n", syn_threads);
DEBUG_FFLUSH();
for (a = 0; a < syn_threads; a++) {
pars[a].cutoff = cutoff;
pars[a].target_sums = target_sums;
pars[a].window_sums = window_sums;
pars[a].wl = wl;
pars[a].N = maxPerPos;
pars[a].best = NULL; /* getCollocators() allocates its own result array */
pars[a].threshold = threshold;
pars[a].from = a;
pars[a].upto = a + 1;
pthread_create(&pt[a], NULL, getCollocators, (void *)&pars[a]);
}
DEBUG_PRINTF("Waiting for syn threads to join\n");
DEBUG_FFLUSH();
for (a = 0; a < syn_threads; a++) pthread_join(pt[a], (void *)&syn_nbs[a]);
DEBUG_PRINTF("Syn threads joint\n");
DEBUG_FFLUSH();
result = malloc((maxPerPos > 0 ? maxPerPos : 1) * (max_w + 96) * syn_threads + 16);
char *p = (char *) result;
*p = 0;
if (strcmp(format, "tsv") == 0) {
for (a = syn_threads - 1; a >= 0; a--) {
if (syn_nbs[a] == NULL) continue;
for (b = 0; b < syn_nbs[a]->length; b++, entries++) {
p += sprintf(p, "%ld\t%s\t%f\n", syn_nbs[a]->best[b].position, &vocab[syn_nbs[a]->best[b].wordi * max_w], syn_nbs[a]->best[b].activation);
}
}
} else {
p += sprintf(p, "[");
for (a = syn_threads - 1; a >= 0; a--) {
if (syn_nbs[a] == NULL) continue;
for (b = 0; b < syn_nbs[a]->length; b++, entries++) {
p += sprintf(p, "{\"pos\": %ld, \"word\":\"%s\",\"activation\": %f},\n", syn_nbs[a]->best[b].position, &vocab[syn_nbs[a]->best[b].wordi * max_w], syn_nbs[a]->best[b].activation);
}
}
if (entries > 0)
p -= 2; /* drop the trailing ",\n" */
p += sprintf(p, "\n]");
}
res_sv = newSVpv(result, 0);
free(result);
free(target_sums);
free(window_sums);
free(pt);
free(wl);
for (a = 0; a < syn_threads; a++) free_knn(syn_nbs[a]);
return res_sv;
}
SV *getPosWiseW2VCollocatorsAsTsv(char *word, long maxPerPos, long cutoff, float threshold) {
return getPosWiseW2VCollocators(word, maxPerPos, cutoff, threshold, "tsv");
}
SV *getPosWiseW2VCollocatorsAsJson(char *word, long maxPerPos, long cutoff, float threshold) {
return getPosWiseW2VCollocators(word, maxPerPos, cutoff, threshold, "json");
}
SV *get_neighbours(char *st1, int N, int sort_by, int search_backw, long cutoff, int dedupe, int no_similar_profiles) {
HV *result = newHV();
float *target_sums = NULL;
float *window_sums = NULL;
long a, b, c, d, slice;
knn *para_nbs[MAX_THREADS];
knn *syn_nbs[MAX_THREADS];
knnpars pars[MAX_THREADS];
pthread_t *pt = (pthread_t *)malloc((num_threads + 1) * sizeof(pthread_t));
wordlist *wl = NULL;
int syn_threads = 0;
int para_threads = 0;
for (a = 0; a < MAX_THREADS; a++) para_nbs[a] = syn_nbs[a] = NULL;
if (N > MAX_NEIGHBOURS) N = MAX_NEIGHBOURS;
if (N < 1) N = 1;
/* Every paradigmatic thread fills its own slice of N entries, and the
syntagmatic part below works on the first MAX_NEIGHBOURS of the same
array. How many paradigmatic threads there are is only known further
down, so the array is sized for the maximum. */
collocator *best = NULL;
long best_entries = (long)N * num_threads;
if (best_entries < MAX_NEIGHBOURS) best_entries = MAX_NEIGHBOURS;
posix_memalign((void **)&best, 128, best_entries * sizeof(collocator));
memset(best, 0, best_entries * sizeof(collocator));
if (cutoff < 1 || cutoff > words)
cutoff = words;
wl = getTargetWords(st1, search_backw);
if (wl == NULL)
goto end;
/* Tell the caller which words the query vector was built from and which
tokens of the query are unknown, so that "König - Mannn + Frau" is not
silently answered as "König + Frau". */
{
SV *added = newSVpv("", 0);
SV *unknown = newSVpv(wl->oov, 0);
for (a = 0; a < wl->length; a++) {
if (wl->sep[a] == '-') continue;
if (SvCUR(added) > 0) sv_catpvn(added, " ", 1);
sv_catpv(added, &vocab[wl->wordi[a] * max_w]);
}
if (latin_enc == 0) {
SvUTF8_on(added);
SvUTF8_on(unknown);
}
hv_store(result, "added", strlen("added"), added, 0);
hv_store(result, "unknown", strlen("unknown"), unknown, 0);
hv_store(result, "operands", strlen("operands"), newSViv(wl->length), 0);
}
if (wl->length < 1)
goto end;
syn_threads = (M2 ? window * 2 : 0);
para_threads = (no_similar_profiles ? 0 : num_threads - syn_threads);
slice = (para_threads > 0 ? cutoff / para_threads : cutoff);
a = posix_memalign((void **)&target_sums, 128, cutoff * sizeof(float));
memset(target_sums, 0, cutoff * sizeof(float));
DEBUG_PRINTF("Starting %d threads for paradigmatic search\n", para_threads);
DEBUG_FFLUSH();
for (a = 0; a < para_threads; a++) {
pars[a].cutoff = cutoff;
pars[a].token = st1;
pars[a].wl = wl;
pars[a].N = N;
pars[a].best = &best[N * a];
if (merge_words == 0 || search_backw == 0) {
pars[a].from = a * slice;
pars[a].upto = ((a + 1) * slice > cutoff ? cutoff : (a + 1) * slice);
} else {
pars[a].from = merge_words + a * slice;
pars[a].upto = merge_words + ((a + 1) * slice > cutoff ? cutoff : (a + 1) * slice);
}
DEBUG_PRINTF("From: %ld, Upto: %ld\n", pars[a].from, pars[a].upto);
pthread_create(&pt[a], NULL, _get_neighbours, (void *)&pars[a]);
}
if (syn_threads) {
window_sums = new_window_sums();
for (a = 0; a < syn_threads; a++) {
pars[a + para_threads].cutoff = cutoff;
pars[a + para_threads].target_sums = target_sums;
pars[a + para_threads].window_sums = window_sums;
pars[a + para_threads].wl = wl;
pars[a + para_threads].N = N;
pars[a + para_threads].threshold = MIN_RESP;
pars[a + para_threads].from = a;
pars[a + para_threads].upto = a + 1;
pthread_create(&pt[a + para_threads], NULL, getCollocators, (void *)&pars[a + para_threads]);
}
}
DEBUG_PRINTF("Waiting for para threads to join\n");
DEBUG_FFLUSH();
for (a = 0; a < para_threads; a++) pthread_join(pt[a], (void *)&para_nbs[a]);
DEBUG_PRINTF("Para threads joint\n");
DEBUG_FFLUSH();
/* if(!syn_nbs[0]) */
/* goto end; */
qsort(best, N * para_threads, sizeof(collocator), cmp_activation);
long long chosen[MAX_NEIGHBOURS];
DEBUG_PRINTF("N: %d\n", N);
AV *array = newAV();
int i, j;
int l1_words = 0, l2_words = 0;
for (a = 0, i = 0; i < N && a < N * para_threads; a++) {
int filtered = 0;
long long c = best[a].wordi;
if (c < 0) /* the threads found fewer candidates than were asked for */
break;
if ((merge_words && dedupe && i > 1) || (!merge_words && dedupe && i > 0)) {
for (j = 0; j < i && !filtered; j++)
if (strcasestr(&vocab[c * max_w], &vocab[chosen[j] * max_w]) ||
strcasestr(&vocab[chosen[j] * max_w], &vocab[c * max_w])) {
DEBUG_PRINTF("filtering %s %s\n", &vocab[chosen[j] * max_w], &vocab[c * max_w]);
filtered = 1;
}
if (filtered)
continue;
}
if (0 && merge_words > 0) {
if (c >= merge_words) {
if (l1_words > N / 2)
continue;
else
l1_words++;
} else {
if (l2_words > N / 2)
continue;
else
l2_words++;
}
}
// printf("%s l1:%d l2:%d i:%d a:%ld\n", &vocab[c * max_w], l1_words, l2_words, i, a);
// fflush(stdout);
HV *hash = newHV();
SV *word = newSVpvf(&vocab[c * max_w], 0);
chosen[i] = c;
if (latin_enc == 0) SvUTF8_on(word);
fflush(stdout);
hv_store(hash, "word", strlen("word"), word, 0);
hv_store(hash, "dist", strlen("dist"), newSVnv(best[a].activation), 0);
hv_store(hash, "rank", strlen("rank"), newSVuv(best[a].wordi), 0);
AV *vector = newAV();
for (b = 0; b < size; b++) {
av_push(vector, newSVnv(M[b + best[a].wordi * size]));
}
hv_store(hash, "vector", strlen("vector"), newRV_noinc((SV *)vector), 0);
av_push(array, newRV_noinc((SV *)hash));
i++;
}
hv_store(result, "paradigmatic", strlen("paradigmatic"), newRV_noinc((SV *)array), 0);
for (b = 0; b < MAX_NEIGHBOURS; b++) {
best[b].wordi = -1L;
best[b].activation = 0;
best[b].raw = 0;
best[b].max_raw = 0;
best[b].probability = 0;
best[b].position = 0;
best[b].activation_sum = 0;
memset(best[b].heat, 0, sizeof(float) * 16);
}
float total_activation = 0;
if (syn_threads) {
DEBUG_PRINTF("Waiting for syn threads to join\n");
DEBUG_FFLUSH();
for (a = 0; a < syn_threads; a++) pthread_join(pt[a + para_threads], (void *)&syn_nbs[a]);
for (a = 0; a <= syn_threads; a++) {
if (a == window) continue;
total_activation += window_sums[a];
DEBUG_PRINTF("window pos: %ld, sum: %f\n", a, window_sums[a]);
}
DEBUG_PRINTF("syn threads joint\n");
DEBUG_FFLUSH();
for (b = 0; b < syn_nbs[0]->length; b++) {
memcpy(best + b, &syn_nbs[0]->best[b], sizeof(collocator));
best[b].position = -1; // syn_nbs[0]->pos[b];
best[b].activation_sum = target_sums[syn_nbs[0]->best[b].wordi];
best[b].max_activation = 0.0;
best[b].max_raw = 0.0;
best[b].average = 0.0;
best[b].probability = 0.0;
best[b].cprobability = syn_nbs[0]->best[b].cprobability;
memset(best[b].heat, 0, sizeof(float) * 16);
}
float best_window_sum[MAX_NEIGHBOURS];
int found_index = 0, i = 0, w;
for (a = 0; a < syn_threads; a++) {
for (b = 0; b < syn_nbs[a]->length; b++) {
for (i = 0; i < found_index; i++)
if (best[i].wordi == syn_nbs[a]->best[b].wordi)
break;
if (i >= found_index) {
best[found_index].max_activation = 0.0;
best[found_index].max_raw = 0.0;
best[found_index].average = 0.0;
best[found_index].probability = 0.0;
memset(best[found_index].heat, 0, sizeof(float) * 16);
best[found_index].cprobability = syn_nbs[a]->best[b].cprobability;
best[found_index].activation_sum = target_sums[syn_nbs[a]->best[b].wordi]; // syn_nbs[a]->best[b].activation_sum;
best[found_index++].wordi = syn_nbs[a]->best[b].wordi;
// printf("found: %s\n", &vocab[syn_nbs[a]->index[b] * max_w]);
}
}
}
sort_by = 0; // ALWAYS AUTO-FOCUS
if (sort_by != 1 && sort_by != 2) { // sort by auto focus mean
DEBUG_PRINTF("window: %d - syn_threads: %d, %d\n", window, syn_threads, (1 << syn_threads) - 1);
int wpos;
int bits_set = 0;
for (i = 0; i < found_index; i++) {
best[i].activation = best[i].probability = best[i].average = best[i].cprobability_sum = 0;
for (w = 1; w < (1 << syn_threads); w++) { // loop through all possible windows
float word_window_sum = 0, word_window_average = 0, word_cprobability_sum = 0, word_activation_sum = 0, total_window_sum = 0;
bits_set = 0;
for (a = 0; a < syn_threads; a++) {
if ((1 << a) & w) {
wpos = (a >= window ? a + 1 : a);
total_window_sum += window_sums[wpos];
}
}
// printf("%d window-sum %f\n", w, total_window_sum);
for (a = 0; a < syn_threads; a++) {
if ((1 << a) & w) {
wpos = (a >= window ? a + 1 : a);
bits_set++;
for (b = 0; b < syn_nbs[a]->length; b++)
if (best[i].wordi == syn_nbs[a]->best[b].wordi) {
// float acti = syn_nbs[a]->best[b].activation / total_window_sum;
// word_window_sum += syn_nbs[a]->dist[b] * syn_nbs[a]->norm[b]; // / window_sums[wpos]; // syn_nbs[a]->norm[b];
// word_window_sum += syn_nbs[a]->norm[b]; // / window_sums[wpos]; // syn_nbs[a]->norm[b];
// word_window_sum = (word_window_sum + syn_nbs[a]->norm[b]) - (word_window_sum * syn_nbs[a]->norm[b]); // syn_nbs[a]->norm[b];
word_window_sum += syn_nbs[a]->best[b].activation; // / window_sums[wpos]; // syn_nbs[a]->norm[b];
// word_window_sum += acti - (word_window_sum * acti); syn_nbs[a]->best[b].activation; // / window_sums[wpos]; // syn_nbs[a]->norm[b];
word_window_average += syn_nbs[a]->best[b].activation; // - word_window_average * syn_nbs[a]->best[b].activation; // conormalied activation sum
word_cprobability_sum += syn_nbs[a]->best[b].cprobability - word_cprobability_sum * syn_nbs[a]->best[b].cprobability; // conormalied column probability sum
word_activation_sum += syn_nbs[a]->best[b].activation;
if (syn_nbs[a]->best[b].activation > best[i].max_activation)
best[i].max_activation = syn_nbs[a]->best[b].activation;
if (syn_nbs[a]->best[b].raw > best[i].max_raw)
best[i].max_raw = syn_nbs[a]->best[b].raw;
if (syn_nbs[a]->best[b].activation > best[i].heat[wpos])
best[i].heat[wpos] = syn_nbs[a]->best[b].activation;
}
}
}
if (bits_set) {
word_window_average /= bits_set;
// word_activation_sum /= bits_set;
// word_window_sum /= bits_set;
}
word_window_sum /= total_window_sum;
if (word_window_sum > best[i].probability) {
// best[i].position = w;
best[i].probability = word_window_sum;
}
if (word_cprobability_sum > best[i].cprobability_sum) {
best[i].position = w;
best[i].cprobability_sum = word_cprobability_sum;
}
best[i].average = word_window_average;
// best[i].activation = word_activation_sum;
}
}
qsort(best, found_index, sizeof(collocator), cmp_probability);
// for(i=0; i < found_index; i++) {
// printf("found: %s - sum: %f - window: %d\n", &vocab[best[i].wordi * max_w], best[i].activation, best[i].position);
// }
} else if (sort_by == 1) { // responsiveness any window position
int wpos;
for (i = 0; i < found_index; i++) {
float word_window_sum = 0, word_activation_sum = 0, total_window_sum = 0;
for (a = 0; a < syn_threads; a++) {
wpos = (a >= window ? a + 1 : a);
for (b = 0; b < syn_nbs[a]->length; b++)
if (best[i].wordi == syn_nbs[a]->best[b].wordi) {
best[i].probability += syn_nbs[a]->best[b].probability;
if (syn_nbs[a]->best[b].activation > 0.25)
best[i].position |= 1 << wpos;
if (syn_nbs[a]->best[b].activation > best[i].activation) {
best[i].activation = syn_nbs[a]->best[b].activation;
}
}
}
}
qsort(best, found_index, sizeof(collocator), cmp_activation);
} else if (sort_by == 2) { // single window position
for (a = 1; a < syn_threads; a++) {
for (b = 0; b < syn_nbs[a]->length; b++) {
for (c = 0; c < MAX_NEIGHBOURS; c++) {
if (syn_nbs[a]->best[b].activation > best[c].activation) {
for (d = MAX_NEIGHBOURS - 1; d > c; d--) {
memmove(best + d, best + d - 1, sizeof(collocator));
}
memcpy(best + c, &syn_nbs[a]->best[b], sizeof(collocator));
best[c].position = 1 << (-syn_nbs[a]->best[b].position + window - (syn_nbs[a]->best[b].position < 0 ? 1 : 0));
break;
}
}
}
}
} else { // sort by mean p
for (a = 1; a < syn_threads; a++) {
for (b = 0; b < syn_nbs[a]->length; b++) {
for (c = 0; c < MAX_NEIGHBOURS; c++) {
if (target_sums[syn_nbs[a]->best[b].wordi] > best[c].activation_sum) {
for (d = MAX_NEIGHBOURS - 1; d > c; d--) {
memmove(best + d, best + d - 1, sizeof(collocator));
}
memcpy(best + c, &syn_nbs[a]->best[b], sizeof(collocator));
best[c].position = (1 << 2 * window) - 1; // syn_nbs[a]->pos[b];
best[c].activation_sum = target_sums[syn_nbs[a]->best[b].wordi];
break;
}
}
}
}
}
array = newAV();
for (a = 0, i = 0; a < MAX_NEIGHBOURS && best[a].wordi >= 0; a++) {
long long c = best[a].wordi;
/*
if (dedupe) {
int filtered=0;
for (j=0; j<i; j++)
if (strcasestr(&vocab[c * max_w], chosen[j]) ||
strcasestr(chosen[j], &vocab[c * max_w])) {
printf("filtering %s %s\n", chosen[j], &vocab[c * max_w]);
filtered = 1;
}
if(filtered)
continue;
}
*/
chosen[i++] = c;
HV *hash = newHV();
SV *word = newSVpvf(&vocab[best[a].wordi * max_w], 0);
AV *heat = newAV();
if (latin_enc == 0) SvUTF8_on(word);
hv_store(hash, "word", strlen("word"), word, 0);
hv_store(hash, "rank", strlen("rank"), newSVuv(best[a].wordi), 0);
hv_store(hash, "average", strlen("average"), newSVnv(best[a].average), 0);
hv_store(hash, "prob", strlen("prob"), newSVnv(best[a].probability), 0);
hv_store(hash, "cprob", strlen("cprob"), newSVnv(best[a].cprobability_sum), 0);
hv_store(hash, "dot", strlen("dot"), newSVnv(best[a].max_raw), 0);
hv_store(hash, "max", strlen("max"), newSVnv(best[a].max_activation), 0); // newSVnv(target_sums[best[a].wordi]), 0);
hv_store(hash, "overall", strlen("overall"), newSVnv(best[a].activation_sum / total_activation), 0); // newSVnv(target_sums[best[a].wordi]), 0);
hv_store(hash, "pos", strlen("pos"), newSVnv(best[a].position), 0);
best[a].heat[5] = 0;
for (i = 10; i >= 0; i--) av_push(heat, newSVnv(best[a].heat[i]));
hv_store(hash, "heat", strlen("heat"), newRV_noinc((SV *)heat), 0);
av_push(array, newRV_noinc((SV *)hash));
}
hv_store(result, "syntagmatic", strlen("syntagmatic"), newRV_noinc((SV *)array), 0);
}
end:
free(best);
free(target_sums);
free(window_sums);
free(pt);
free(wl);
for (a = 0; a < MAX_THREADS; a++) {
free_knn(para_nbs[a]);
free_knn(syn_nbs[a]);
}
return newRV_noinc((SV *)result);
}
int dump_vecs(char *fname) {
long i, j;
FILE *f;
/* if(words>100000)
words=100000;
*/
if ((f = fopen(fname, "w")) == NULL) {
fprintf(stderr, "cannot open %s for writing\n", fname);
return (-1);
}
fprintf(f, "%lld %lld\n", words, size);
for (i = 0; i < words; i++) {
fprintf(f, "%s ", &vocab[i * max_w]);
for (j = 0; j < size - 1; j++)
fprintf(f, "%f ", M[i * size + j]);
fprintf(f, "%f\n", M[i * size + j]);
}
fclose(f);
return (0);
}
int dump_for_numpy(char *fname) {
long i, j;
FILE *f;
int max = words; // 300000;
if ((f = fopen(fname, "w")) == NULL) {
fprintf(stderr, "cannot open %s for writing\n", fname);
return (-1);
}
for (i = 0; i < max; i++) {
for (j = 0; j < size - 1; j++)
fprintf(f, "%f\t", M[i * size + j]);
fprintf(f, "%f\n", M[i * size + j]);
printf("%s\r\n", &vocab[i * max_w]);
}
if (merged_end > 0) {
for (i = 0; i < max; i++) {
for (j = 0; j < size - 1; j++)
fprintf(f, "%f\t", M[(merged_end + i) * size + j]);
fprintf(f, "%f\n", M[(merged_end + i) * size + j]);
printf("_%s\r\n", &vocab[i * max_w]);
}
}
fclose(f);
return (0);
}
unsigned long getVocabSize() {
return (unsigned long) words;
}
/* First rank of the primary model in the merged vocabulary, 0 if no second
model was merged in. mergeVectors() puts the merged in model at ranks
[0, merged_end) and the primary model - the one the collocator db belongs
to - at [merged_end, words). */
long getMergedEnd() {
return (long) merged_end;
}