| <!DOCTYPE html> |
| <html> |
| <head> |
| <% my $plain_title = $title; $plain_title=~s/<[^>]+>//g; %> |
| %# Whether the query was answered at all: not found, and an expression whose |
| %# operands cancel each other out, both leave the first list empty. |
| % my $have_results = ($lists && @$lists > 0 && $lists->[0] && @{$lists->[0]} > 0); |
| %# The count based collocators come out of the co-occurrence database, which is |
| %# keyed by one node word, so they have no answer for a query of several |
| %# operands. The predictive ones do, see getCollocators(). |
| % my $single_word = ($operands == 1); |
| <title><%= $plain_title %>:<%= $word %> · IDS word vector analysis</title> |
| %# Result pages are computed for one word each and there is one for every |
| %# word of the vocabulary, so they are not something to index or to follow. |
| % if (stash 'noindex') { |
| <meta name="robots" content="noindex, nofollow"> |
| % } |
| <link rel="stylesheet" href="//code.jquery.com/ui/1.12.1/themes/base/jquery-ui.css"> |
| <link href="https://fonts.googleapis.com/css?family=Lato|Roboto+Condensed" rel="stylesheet"> |
| <script src="https://code.jquery.com/jquery-latest.min.js"></script> |
| <script src = "https://cdn.datatables.net/1.13.10/js/jquery.dataTables.min.js"></script> |
| <script src = "https://cdn.datatables.net/fixedcolumns/3.2.5/js/dataTables.fixedColumns.min.js"></script> |
| <script src = "https://cdn.datatables.net/plug-ins/1.13.10/sorting/scientific.js"></script> |
| <script src='https://cdnjs.cloudflare.com/ajax/libs/mathjax/2.7.5/MathJax.js?config=TeX-MML-AM_CHTML'></script> |
| <script src="https://cdn.datatables.net/buttons/2.3.6/js/dataTables.buttons.min.js"></script> |
| <script src="https://cdn.datatables.net/buttons/2.3.6/js/buttons.print.min.js"></script> |
| <script src="https://cdnjs.cloudflare.com/ajax/libs/jszip/3.1.3/jszip.min.js"></script> |
| <script src="https://cdn.datatables.net/buttons/2.3.6/js/buttons.html5.min.js"></script> |
| |
| <script src="https://cdnjs.cloudflare.com/ajax/libs/ical.js/1.5.0/ical.js" |
| integrity="sha512-UxWd2RMDGpYwYuDeU2Fs+nd51gw4ZLQY0D9wWZ3NuanNSk6QQmWCZ6C+rvI0lJ/b/nZtBQ8RVESA9DN9r65dUg==" |
| crossorigin="anonymous" referrerpolicy="no-referrer"></script> |
| <script |
| src="https://cdnjs.cloudflare.com/ajax/libs/moment.js/2.29.4/moment-with-locales.min.js" |
| integrity="sha512-42PE0rd+wZ2hNXftlM78BSehIGzezNeQuzihiBCvUEB3CVxHvsShF86wBWwQORNxNINlBPuq7rG4WWhNiTVHFg==" |
| crossorigin="anonymous" referrerpolicy="no-referrer"></script> |
| <link rel="stylesheet" href="https://cdn.datatables.net/1.13.10/css/jquery.dataTables.min.css"> |
| <link rel="stylesheet" href="/derekovecs/css/derekovecs.css"> |
| <script |
| src="https://code.jquery.com/ui/1.12.1/jquery-ui.min.js" |
| integrity="sha256-VazP97ZCwtekAsvgPBSUwPFKdrwD3unUfSGVYrahUqU=" |
| crossorigin="anonymous"></script> |
| <script> |
| MathJax.Hub.Config({ |
| config: ["MMLorHTML.js"], |
| jax: ["input/TeX","input/MathML","output/HTML-CSS","output/NativeMML", "output/PreviewHTML"], |
| extensions: ["tex2jax.js","mml2jax.js","MathMenu.js","MathZoom.js", "fast-preview.js", "AssistiveMML.js", "a11y/accessibility-menu.js"], |
| TeX: { |
| extensions: ["AMSmath.js","AMSsymbols.js","noErrors.js","noUndefined.js"] |
| } |
| }); |
| var urlParams = new URLSearchParams(window.location.search); |
| var currentWords = urlParams.get("word"); |
| % use Mojo::ByteStream 'b'; |
| // The vocabulary entries the query vector was built from, as the server |
| // resolved them. A query can be a vector expression, "König - Mann + |
| // Frau", and the subtracted operands are deliberately not in here: they |
| // are not marked in the maps, and not part of the KorAP queries built |
| // from the result. |
| var searchedWords = <%= b(Mojo::JSON::to_json($added // '')) %>; |
| var targetWords = " " + searchedWords + " "; |
| var CIIsearchWords = (searchedWords.includes(" ") ? '('+searchedWords.replace(/\s+/g, " oder ")+')' : searchedWords); |
| var collocatorTable = null; |
| var classicCollocatorTable = null; |
| var singleWord = <%= $single_word ? 'true' : 'false' %>; |
| var plainTitle ="<%= $plain_title %>" |
| var korapPath="/"; |
| if (plainTitle.match(/-en/)) { |
| korapPath="/instance/english"; |
| } |
| var korapVC=""; |
| if (plainTitle.match(/WUDD/i)) { |
| korapVC="&collection=corpusSigle+%3D+%2FW%5BDU%5DD17%2F"; |
| korapPath="/instance/test"; |
| } |
| |
| $(document).ready(function() { |
| insertCalendarEvents(document.getElementById('downtimes'), '/derekovecs/getDowntimeCalendar', 7); |
| |
| $('#firstable').hide(); |
| //Set up a callback to hear back when MathJax is done rendering the equations |
| // it finds |
| $('#ccd').load( |
| '@Url.Action("ActionResultMethod","ControllerName",{controller parameters})', |
| function () { |
| MathJax.Hub.Queue( |
| ["Typeset",MathJax.Hub,"ccd"], |
| function () { |
| $("#mi_tt").attr("title",$("#pmi_ttt").html()); |
| $("#lfmd_tt").attr("title",$("#lfmd_ttt").html()); |
| $("#md_tt").attr("title",$("#md_ttt").html()); |
| $("#npmi_tt").attr("title",$("#npmi_ttt").html()); |
| $("#ll_tt").attr("title",$("#ll_ttt").html()); |
| $("#logdice_tt").attr("title",$("#logdice_ttt").html()); |
| $("#logdiceaf_tt").attr("title",$("#logdiceaf_ttt").html()); |
| } |
| ); |
| }); |
| |
| //set things up so that we can shove raw html into what is shown in the tooltip; |
| // in this case, we will have already put into the title attribute the html that |
| // contains the MathJax rendered equations (via what we do in the callback). |
| $(function () { |
| $(document).tooltip({ |
| content: function () { |
| return $(this).prop('title'); |
| } |
| }); |
| }); |
| |
| $("input").bind("keydown", function(event) { |
| // track enter key |
| var keycode = (event.keyCode ? event.keyCode : (event.which ? event.which : event.charCode)); |
| if (keycode == 13) { // keycode for enter key |
| // force the 'Enter Key' to implicitly click the Update button |
| document.getElementById('SEARCH').click(); |
| return false; |
| } else { |
| return true; |
| }}); |
| |
| var collocatorTable_activated = false; |
| $( "#tabs" ).on( "tabsactivate", function( event, ui ) { |
| if (localStorage) localStorage['tab'] = ui.newTab.index(); |
| if(ui.newTab.index() == 3 && !collocatorTable_activated && collocatorTable) { |
| if (classicCollocatorTable) classicCollocatorTable.columns.adjust(); |
| collocatorTable.columns.adjust(); |
| collocatorTable_activated = true; |
| } |
| }); |
| |
| $(function(){ |
| $("#SEARCH").click(function() { |
| window.open($(location).attr('pathname')+'?'+$('form').serialize(), "_self"); |
| }); |
| }); |
| |
| function changeCharColor(txt, heat, word) { |
| var newText = ""; |
| for (var i=0, l=txt.length; i<l; i++) { |
| newText += (i == 5 ? txt.charAt(i) : "<a href=\"<%= $korap_url %>" + korapPath + '/?ql=cosmas2&q=' + |
| CIIsearchWords + ' /' + (i > 5? '%2B' : '-') + 'w' + |
| Math.abs(i-5) + ':' + Math.abs(i-5) + ',s0 ' + word + korapVC + |
| '" target="korap"><span style="background-color:' + |
| getHeatColor(heat[i]/maxHeat)+'">'+txt.charAt(i)+'</span></a>'); |
| } |
| return newText; |
| } |
| |
| function getHeatColor(value) { |
| var hue=((1-value)*120).toString(10); |
| return ["hsl(",hue,",90%,70%)"].join(""); |
| } |
| |
| function bitvec2window(n, heat, word) { |
| var str = n.toString(2).padStart(10, "0") |
| .replace(/^([0-9]{5})/, '$1x') |
| .replace(/0/g, '·') |
| .replace(/1/g, '+'); |
| return changeCharColor(str, heat, word); |
| } |
| |
| var paraResults = <%= b(Mojo::JSON::to_json($lists)) %>; |
| var urlprefix = new URLSearchParams(window.location.search); |
| if (paraResults.length > 0 && paraResults[0] != null && paraResults[0].length > 0) { |
| var nvecs = [], |
| nwords = [], |
| nranks = [], |
| nmarked = []; |
| for(var i = 0; i < paraResults.length; i++) { |
| nwords = nwords.concat(paraResults[i].map(function(a){return a.word;})); |
| nvecs = nvecs.concat(paraResults[i].map(function(a){return a.vector;})); |
| nranks = nranks.concat(paraResults[i].map(function(a){return a.rank;})); |
| nmarked = nmarked.concat(paraResults[i].map(function(a){return a.marked;})); |
| } |
| showMap({target: targetWords, mergedEnd: <%= $mergedEnd %>, words: nwords, vecs: nvecs, ranks: nranks, marked: nmarked} ); |
| var t = $('#firsttable').DataTable({ |
| data: [].concat.apply([], paraResults), |
| "sScrollY": "780px", |
| "bScrollCollapse": true, |
| "bPaginate": false, |
| "bJQueryUI": true, |
| "dom": '<"top">rt<"bottom"flpB><"clear">', |
| buttons: [ |
| { extend: 'copyHtml5', text: '<i class="fa fa-files-o"></i>', titleAttr: 'Copy' }, |
| { extend: 'csvHtml5', text: '<i class="fa fa-file-text-o"></i>', titleAttr: 'CSV', filename: currentWords+'_para_neighbours' }, |
| { extend: 'excelHtml5', text: '<i class="fa fa-file-excel-o"></i>', titleAttr: 'Excel', filename: currentWords+'_para_neighbours', exportOptions: { columns: ':visible' } , |
| } |
| ], |
| "initComplete":function(settings, json) { |
| $('td.paradigmator a').on('mousedown', function(e) { |
| return paradigmatorClick(e, paraResults[0][0].word, this.childNodes["0"].textContent); |
| }); |
| }, |
| "columns": [ |
| { "data": "rank", type: "allnumeric" }, |
| { "data": "dist", render: function ( data, type, row ) {return data.toFixed(3) }}, |
| { "data": "word", class: "paradigmator", render: function ( data, type, row ) { |
| urlprefix.set("word", data); return '<a class="' + getMergedClass(row.rank) + '" href="?' + urlprefix + '">' + data + '</a>' |
| }} |
| ], |
| "columnDefs": [ |
| { className: "dt-right", "targets": [0,1] }, |
| { "searchable": false, |
| "orderable": false, |
| "targets": 0 |
| }, |
| { "orderSequence": [ "desc" ], "targets": [ 1 ] }, |
| { "orderSequence": [ "asc", "desc" ], "targets": [ 2 ] }, |
| ], |
| "oLanguage": { |
| "sSearch": "Filter: " |
| }, |
| "order": [[ 1, 'desc' ]], |
| } ); |
| |
| t.on( 'order.dt search.dt', function () { |
| t.column(0, {order:'applied'}).nodes().each( function (cell, i) { |
| cell.innerHTML = i+1; |
| } ); |
| } ).draw(); |
| |
| $( "#first" ).clone().prependTo( "#tabs-2" ); |
| |
| } |
| |
| |
| var collocatorData = <%= b(Mojo::JSON::to_json($collocators)) %>; |
| var maxHeat; // = Math.max.apply(Math,collocatorData.map(function(o){return o.cprob;})) |
| |
| if(typeof data !== 'undefined' && data.mergedEnd > 0) { |
| vocabDistanceTable = makeVocabDistanceTable("#vocabdistt", baseURL); |
| } |
| |
| if (collocatorData != null) { |
| maxHeat = Math.max.apply(Math,collocatorData.map(function(o){return Math.max.apply(Math,o.heat);})) |
| collocatorTable = $('#secondtable').DataTable({ |
| data: collocatorData, |
| "sScrollY": "780px", |
| "bScrollCollapse": true, |
| "bPaginate": false, |
| "bJQueryUI": true, |
| "dom": '<"top">rt<"bottom"flpB><"clear">', |
| buttons: [ |
| { extend: 'copyHtml5', text: '<i class="fa fa-files-o"></i>', titleAttr: 'Copy' }, |
| { extend: 'csvHtml5', text: '<i class="fa fa-file-text-o"></i>', titleAttr: 'CSV', filename: currentWords+'_syn_neighbours' }, |
| { extend: 'excelHtml5', text: '<i class="fa fa-file-excel-o"></i>', titleAttr: 'Excel', filename: currentWords+'_syn_neighbours', exportOptions: { columns: ':visible' } , |
| } |
| ], |
| "columns": [ |
| { "data": "rank", type: "allnumeric" }, |
| { "data": "pos", width: "7%", sClass: "dt-center mono compact", render: function ( data, type, row ) {return bitvec2window(data, row.heat, row.word) }}, |
| { "data": "max", render: function ( data, type, row ) {return data.toFixed(3) }}, |
| { "data": "average", render: function ( data, type, row ) {return data.toFixed(3) }}, |
| { "data": "prob", type: "scientific", render: function ( data, type, row ) {return data.toExponential(3) } }, |
| { "data": "cprob", type: "scientific", render: function ( data, type, row ) {return data.toExponential(3) } }, |
| { "data": "overall", type: "scientific", render: function ( data, type, row ) {return data.toExponential(3) } }, |
| { "data": "word", sClass: "collocator" }, |
| { "data": "rank", type: "allnumeric" } |
| ], |
| "columnDefs": [ |
| { className: "dt-right", "targets": [0,2,3,4,5,6] }, |
| { className: "dt-center", "targets": [ 1] }, |
| { "searchable": false, |
| "orderable": false, |
| "targets": [0, 8] |
| }, |
| { "type": "scientific", targets: [2,3,4,5,6] }, |
| { "orderSequence": [ "desc" ], "targets": [ 2, 3, 4, 5, 6 ] }, |
| { "orderSequence": [ "asc", "desc" ], "targets": [ 1, 7 ] }, |
| { "targets": [8], "visible": false } |
| ], |
| "oLanguage": { |
| "sSearch": "Filter: " |
| }, |
| "order": [[ 4, 'desc' ]], |
| } ); |
| $.ajaxSetup({ |
| type: 'POST', |
| timeout: 30000, |
| error: function(xhr) { |
| $('#display_error') |
| .html('Error: ' + xhr.status + ' ' + xhr.statusText); |
| } |
| }); |
| |
| |
| if($('#sprofiles').length) { |
| similarProfileTable = $('#sprofiles').DataTable({ |
| ajax: { |
| method: "GET", |
| url: '/derekovecs/getSimilarProfiles', |
| dataType: 'json', |
| dataSrc: "", |
| timeout: 30000, |
| data: { w: paraResults[0][0].rank } |
| }, |
| "initComplete":function(settings, json){ |
| $('td.paradigmator a').on('mousedown', function(e) { |
| if (e.which === 2) { |
| e.preventDefault(); |
| queryKorAPalternatives(paraResults[0][0].word, this.childNodes["0"].textContent); |
| return false; |
| } |
| }); |
| }, |
| "sScrollY": "780px", |
| "bScrollCollapse": true, |
| "bPaginate": false, |
| "bJQueryUI": true, |
| "dom": '<"top">rt<"bottom"flp><"clear">', |
| "columns": [ |
| { "data": "v", render: function ( data, type, row ) {return data.toFixed(3) }}, |
| { "data": "w", sClass: "paradigmator", render: function ( data, type, row ) {urlprefix.set("word", data); return '<a href="?' + urlprefix + '">' + data + '</a>' } } |
| ], |
| "columnDefs": [ |
| { className: "dt-right", "targets": [0] }, |
| ], |
| "oLanguage": { |
| "sSearch": "Filter: " |
| }, |
| "order": [[ 0, 'desc' ]], |
| }); |
| } |
| |
| // var filterQuot = /(^quot?=[A-Z])|(quot$)/g; |
| var filterQuot = /^quot/; |
| var ccResult; |
| var baseURL = window.location.pathname.replace(/[/]$/, '') |
| // Only for a one word query: the node of a count based profile is a |
| // word, and paraResults[0][0] is the nearest word to the query |
| // vector, which is not what was asked for. |
| if (singleWord) { |
| classicCollocatorTable = makeClassicCollocatorTable('#classicoloctable', baseURL, paraResults[0][0].rank) |
| |
| $('#show-details').change(function (e) { |
| var columns = classicCollocatorTable.columns(".detail"); |
| if(this.checked) { |
| columns.visible(true); |
| $("#ccd").css('width', 'auto'); |
| } else { |
| columns.visible(false); |
| $("#ccd").css('width', '680px'); |
| } |
| classicCollocatorTable.columns.adjust().draw(); |
| } ); |
| } |
| |
| $("td.collocator").click(function(){ |
| queryKorAPCII(this.textContent + " /w1:5,s0 " + CIIsearchWords); |
| }); |
| |
| collocatorTable.on( 'order.dt search.dt', function () { |
| collocatorTable.column(0, {order:'applied'}).nodes().each( function (cell, i) { |
| cell.innerHTML = i+1; |
| } ); |
| }).draw(); |
| } |
| |
| if (localStorage && !window.location.hash) { // let's not crash if some user has IE7 |
| var index = parseInt(localStorage['tab']||'0'); |
| $("#tabs").tabs({ active: index }); |
| } |
| $("#tabs").css("visibility", "visible"); // now we can show the tabs |
| }); |
| |
| $(function(){ |
| $("#dropdownoptions").dialog({ |
| title: "<%= loc 'Options' %>", |
| autoOpen: false, |
| modal: false, |
| draggable: false, |
| height: "auto", |
| width: "auto", |
| resizable: false, |
| buttons: { |
| "<%= loc 'Cancel'%>": function() { |
| $( this ).dialog( "close" ); |
| }, |
| "<%= loc 'Apply' %>": function() { |
| window.open($(location).attr('pathname')+'?'+$('form').serialize(), "_self"); |
| } |
| } |
| }); |
| }); |
| |
| $(function(){ |
| $("#showoptions").click(function(){ |
| $("#dropdownoptions").dialog("open"); |
| var target = $(this); |
| $("#dropdownoptions").dialog("widget").position({ |
| my: 'left bottom', |
| at: 'left bottom', |
| of: target |
| }); |
| }); |
| }); |
| |
| $( function() { |
| $( "#no_iterations" ).spinner({ |
| spin: function( event, ui ) { |
| if ( ui.value < 1000 ) { |
| $( this ).spinner( "value", 1000 ); |
| return false; |
| } else if ( ui.value > 10000 ) { |
| $( this ).spinner( "value", 10000 ); |
| return false; |
| } |
| } |
| }); |
| } ); |
| |
| $( function() { |
| $( "#neighbours" ).spinner({ |
| spin: function( event, ui ) { |
| if ( ui.value < 0 ) { |
| $( this ).spinner( "value", 0 ); |
| return false; |
| } else if ( ui.value > 200 ) { |
| $( this ).spinner( "value", 200 ); |
| return false; |
| } |
| } |
| }); |
| } ); |
| |
| $( function() { |
| $( "#cutoff" ).spinner({ |
| spin: function( event, ui ) { |
| if ( ui.value < 100000 ) { |
| $( this ).spinner( "value", 100000 ); |
| return false; |
| } else if ( ui.value > 2000000 ) { |
| $( this ).spinner( "value", 2000000 ); |
| return false; |
| } |
| } |
| }); |
| } ); |
| |
| $( function() { |
| $( "#tabs" ).tabs().addClass('tabs-min'); |
| } ); |
| |
| $( function() { |
| $( ".controlgroup-vertical" ).controlgroup({ |
| "direction": "vertical" |
| }); |
| } ); |
| |
| $(function() { |
| $( document ).tooltip({ |
| content: function() { |
| return $(this).attr('title'); |
| }} |
| ) |
| }); |
| |
| $(function () { |
| $(document).tooltip({ |
| content: function () { |
| return $(this).prop('title'); |
| }, |
| show: null, |
| close: function (event, ui) { |
| ui.tooltip.hover( |
| function () { |
| $(this).stop(true).fadeTo(400, 1); |
| }, |
| function () { |
| $(this).fadeOut("400", function () { |
| $(this).remove(); |
| }) |
| }); |
| } |
| }); |
| }); |
| </script> |
| <script src="//d3js.org/d3.v3.min.js" charset="utf-8"></script> |
| <script src="/derekovecs/js/calendar-events.js"></script> |
| <script src="/derekovecs/js/tsne.js"></script> |
| <script src="/derekovecs/js/som.js"></script> |
| <script src="/derekovecs/js/labeler.js"></script> |
| <script src="/derekovecs/js/derekovecs.js"></script> |
| <script> |
| |
| var opt = {epsilon: <%= $epsilon %>, perplexity: <%= $perplexity %>}, |
| mapWidth = 800, // width map |
| mapHeight = 800, |
| jitterRadius = 7; |
| |
| var T = new tsnejs.tSNE(opt); // create a tSNE instance |
| |
| var Y; |
| |
| var data; |
| var labeler; |
| |
| function applyJitter() { |
| svg.selectAll('.tsnet') |
| .data(labels) |
| .transition() |
| .duration(50) |
| .attr("transform", function(d, i) { |
| T.Y[i][0] = (d.x - mapWidth/2 - tx)/ss/20; |
| T.Y[i][1] = (d.y - mapHeight/2 - ty)/ss/20; |
| return "translate(" + |
| (d.x) + "," + |
| (d.y) + ")"; |
| }); |
| } |
| |
| function updateEmbedding() { |
| var Y = T.getSolution(); |
| svg.selectAll('.tsnet') |
| .data(data.words) |
| .attr("transform", function(d, i) { |
| return "translate(" + |
| ((Y[i][0]*20*ss + tx) + mapWidth/2) + "," + |
| ((Y[i][1]*20*ss + ty) + mapHeight/2) + ")"; }); |
| } |
| |
| var svg; |
| var labels = []; |
| var anchor_array = []; |
| var text; |
| |
| function getMergedClass(i) { |
| if(typeof data !== 'undefined' && i > data.mergedEnd) { |
| return " merged" |
| } else { |
| return ""; |
| } |
| } |
| |
| function getRankTooltip(i) { |
| if(data.mergedEnd) { |
| if(data.ranks[i] < data.mergedEnd) { |
| return "rank: "+i +" "+"freq. rank: "+(data.ranks[i]).toString().replace(/\B(?=(\d{3})+(?!\d))/g, ","); |
| } else { |
| return "rank: "+i +" "+"freq. rank: "+(data.ranks[i]-data.mergedEnd).toString().replace(/\B(?=(\d{3})+(?!\d))/g, ",") + " (merged vocab)"; |
| } |
| } else { |
| return "rank: "+i +" "+"freq. rank: "+data.ranks[i].toString().replace(/\B(?=(\d{3})+(?!\d))/g, ","); |
| } |
| } |
| |
| function drawEmbedding() { |
| var urlprefix = new URLSearchParams(window.location.search); |
| urlprefix.delete("word"); |
| urlprefix.append("word",""); |
| |
| $("#embed").empty(); |
| var div = d3.select("#embed"); |
| |
| // get min and max in each column of Y |
| var Y = T.Y; |
| |
| svg = div.append("svg") // svg is global |
| .attr("width", mapWidth) |
| .attr("height", mapHeight); |
| |
| var g = svg.selectAll(".b") |
| .data(data.words) |
| .enter().append("g") |
| .attr("class", "tsnet"); |
| |
| g.append("a") |
| .attr("xlink:href", function(word) { |
| return "?"+urlprefix+word; }) |
| .attr("class", function(d, i) { |
| var res=""; |
| if(data.marked[i]) { |
| res="marked "; |
| } |
| if(data.target.indexOf(" "+d+" ") >= 0) { |
| res += "target"; |
| } |
| if(data.mergedEnd && data.ranks[i] >= data.mergedEnd) { |
| return res+" merged"; |
| } else { |
| return res; |
| } |
| }) |
| .attr("title", function(d, i) { |
| return getRankTooltip(i); |
| }) |
| .append("text") |
| .attr("text-anchor", "top") |
| .attr("font-size", 12) |
| .text(function(d) { return d; }); |
| |
| g.append("svg:title") |
| .text(function(d, i) { |
| return getRankTooltip(i); |
| }); |
| |
| var zoomListener = d3.behavior.zoom() |
| .scaleExtent([0.1, 10]) |
| .center([0,0]) |
| .on("zoom", zoomHandler); |
| zoomListener(svg); |
| } |
| |
| var tx=0, ty=0; |
| var ss=1; |
| var iter_id=-1; |
| |
| function zoomHandler() { |
| tx = d3.event.translate[0]; |
| ty = d3.event.translate[1]; |
| ss = d3.event.scale; |
| updateEmbedding(); |
| } |
| |
| var stepnum = 0; |
| |
| function stopStep() { |
| clearInterval(iter_id); |
| text = svg.selectAll("text"); |
| |
| // jitter function needs different data and co-ordinate representation |
| labels = d3.range(data.words.length).map(function(i) { |
| var x = (T.Y[i][0]*20*ss + tx) + mapWidth/2; |
| var y = (T.Y[i][1]*20*ss + ty) + mapHeight/2; |
| anchor_array.push({x: x, y: y, r: jitterRadius}); |
| return { |
| x: x, |
| y: y, |
| name: data.words[i] |
| }; |
| }); |
| |
| // get the actual label bounding boxes for the jitter function |
| var index = 0; |
| text.each(function() { |
| labels[index].width = this.getBBox().width; |
| labels[index].height = this.getBBox().height; |
| index += 1; |
| }); |
| |
| |
| // setTimeout(updateEmbedding, 1); |
| // setTimeout( |
| labeler = d3.labeler() |
| .label(labels) |
| .anchor(anchor_array) |
| .width(mapWidth) |
| .height(mapHeight) |
| .update(applyJitter); |
| // .start(1000); |
| |
| iter_id = setInterval(jitterStep, 1); |
| } |
| |
| var jitter_i=0; |
| |
| function jitterStep() { |
| if(jitter_i++ > 100) { |
| clearInterval(iter_id); |
| } else { |
| labeler.start2(10); |
| applyJitter(); |
| } |
| } |
| |
| var last_cost=1000; |
| |
| function step() { |
| var i = T.iter; |
| |
| if(i > <%= $no_iterations %>) { |
| stopStep(); |
| } else { |
| var cost = Math.round(T.step() * 100000) / 100000; // do a few steps |
| $("#cost").html("tsne iteration " + i + ", cost: " + cost.toFixed(5)); |
| if(i % 250 == 0 && cost >= last_cost) { |
| stopStep(); |
| } else { |
| last_cost = cost; |
| updateEmbedding(); |
| } |
| } |
| } |
| |
| function showMap(j) { |
| data=j; |
| T.iter=0; |
| iter_id = -1; |
| last_cost=1000; |
| T.initDataRaw(data.vecs); // init embedding |
| drawEmbedding(); // draw initial embedding |
| |
| if(iter_id >= 0) { |
| clearInterval(iter_id); |
| } |
| //T.debugGrad(); |
| iter_id = setInterval(step, 1); |
| if(true) { // (<%= $show_som %>) { |
| makeSOM(j, <%= $no_iterations %>); |
| } |
| } |
| var queryword; |
| |
| function showCollocatorSOM() { |
| var baseURL = window.location.pathname.replace(/[/]$/, '') |
| if (collocatorTable) { |
| var ctableData = collocatorTable.rows().data(); |
| var nwords = [], |
| nranks = []; |
| for (var i=0; i < ctableData.length && i < 100; i++) { |
| nranks.push(ctableData[i].rank); |
| nwords.push(ctableData[i].word); |
| } |
| $.post(baseURL+'/getVecsByRanks', |
| JSON.stringify(nranks), |
| function(data, status){ |
| showMap({target: targetWords, mergedEnd: <%= $mergedEnd %>, words: nwords, vecs: data, ranks: nranks, marked: Array(100).fill(false)} ); |
| }, 'json'); |
| } |
| } |
| |
| function onload() { |
| queryword = document.getElementById('word'); |
| } |
| |
| function queryKorAP() { |
| window.open("<%= $korap_url %>" + korapPath + '?q='+queryword.value+korapVC, 'KorAP'); |
| } |
| |
| function queryKorAPCII(query) { |
| window.open("<%= $korap_url %>" +korapPath + '?ql=cosmas2&q='+query+korapVC, 'KorAP'); |
| } |
| |
| </script> |
| </head> |
| <body onload="onload()"> <div style="display:none;" id="pmi_ttt">Pointwise mutual information: $$\text{MI}=\text{MI}=\log_2\frac{p(w_1,w_2)}{p(w_1) p(w_2)}$$<p class="citation">Church, K. W. and Hanks, P. (1990): Word association norms, mutual information, and lexicography. Comput. Linguist. 16, 1 (March 1990), 22-29.</p></div> |
| <div style="display:none;" id="md_ttt">Pointwise mutual information squared [1], also called mutual dependency [2]: $$\text{MI}^2=\text{MD}=\log_2\frac{p^2(w_1,w_2)}{p(w_1) p(w_2)}$$<p class="citation">[1] Daille, B. (1994): <a href="http://www.bdaille.com/index.php?option=com_docman&task=doc_download&gid=8&Itemid=">Approche mixte pour l’extraction automatique de terminologie: statistiques lexicales et filtres linguistiques</a>. PhD thesis, Université Paris 7.</p><p class="citation">[2] Thanopoulos, A., Fakotakis, N., Kokkinakis, G. (2002): <a href="http://citeseerx.ist.psu.edu/viewdoc/download?doi=10.1.1.11.8101&rep=rep1&type=pdf">Comparative evaluation of collocation extraction metrics</a>. In: Proc. of LREC 2002: 620–625.</p></div> |
| <div style="display:none;" id="lfmd_ttt">Pointwise mutual information cubed [1], also called log-frequency biased mutual dependency [2]: $$\text{MI}^3=\text{LFMD}=\log_2\frac{p^3(w_1,w_2)}{p(w_1) p(w_2)}$$<p class="citation">[1] Daille, B. (1994): <a href="http://www.bdaille.com/index.php?option=com_docman&task=doc_download&gid=8&Itemid=">Approche mixte pour l’extraction automatique de terminologie: statistiques lexicales et filtres linguistiques</a>. PhD thesis, Université Paris 7.</p><p class="citation">[2] Thanopoulos, A., Fakotakis, N., Kokkinakis, G. (2002): <a href="http://citeseerx.ist.psu.edu/viewdoc/download?doi=10.1.1.11.8101&rep=rep1&type=pdf">Comparative evaluation of collocation extraction metrics</a>. In: Proc. of LREC 2002: 620–625.</p></div> |
| <div style="display:none;" id="npmi_ttt">Normalized pointwise mutual information: $$\frac{\log_2\frac{p(w_1,w_2)}{p(w_1)p(w_2)}}{-\log_2(p(w_1,w_2))}$$<p class="citation">Bouma, Gerlof (2009): <a href="https://svn.spraakdata.gu.se/repos/gerlof/pub/www/Docs/npmi-pfd.pdf">Normalized (pointwise) mutual information in collocation extraction</a>. In Proceedings of GSCL.</p></div> |
| <div style="display:none;" id="ll_ttt">Log-likelihood: $$2\sum_{ij}O_{ij}\log\frac{O_{ij}}{E_{ij}}$$<p class="citation">Dunning, T. (1993): Accurate methods for the statistics of surprise and coincidence. Comput. Linguist. 19, 1 (March 1993), 61-74.</p> |
| <p class="citation">Evert, Stefan (2004): <a href="http://purl.org/stefan.evert/PUB/Evert2004phd.pdf">The Statistics of Word Cooccurrences: Word Pairs and Collocations.</a> PhD dissertation, IMS, University of Stuttgart. Published in 2005, URN urn:nbn:de:bsz:93-opus-23714.</p></div> |
| <div style="display:none;" id="logdice_ttt">Log-Dice: $$14 + \log_2 \frac{2f_{1,2}}{f_1 + f_2}$$<p class="citation">Rychlý, Pavel (2008): <a href="http://www.fi.muni.cz/usr/sojka/download/raslan2008/13.pdf">A lexicographer-friendly association score.</a> In Proceedings of Recent Advances in Slavonic Natural Language Processing, RASLAN, 6–9, 2008</p></div> |
| <div style="display:none;" id="logdiceaf_ttt">Log-Dice using "auto-focus", i.e. the highest value reached by any – not necessarily contiguous – selection \(S\) of positions: $$\max_S \left(14 + \log_2 \frac{2f_{1,2}(S)}{f_2 + f_1|S|}\right)$$ Dividing by the width of the selection makes the measure sensitive to how concentrated a pair is, which is what distinguishes actual collocations from words that merely occur in the same contexts. Note that LDaf is therefore not on the same scale as LD and can be higher or lower than it.</div> |
| <div id="ids_logo"> |
| <a href="http://www.ids-mannheim.de/" target="_blank"><img src="/derekovecs/img/IDS-neu_farbig.svg" alt="Leibniz-Institut für Deutsche Sprache"/></a> |
| </div> |
| <div id="header"> |
| <div id="pagetitle"> |
| <h1>DeReKoVecs</h1> |
| <h2><%== $title %></h2> |
| </div> |
| <div id="options" class="widget"> |
| <form id="queryform"> |
| <input id="word" type="text" name="word" placeholder="<%= loc 'words_to_be_searched' %>" value="<%= $word %>" |
| title="<%= loc 'search_description' %>"/> |
| <input id="SEARCH" type="button" value="<%= loc 'SEARCH' %>"> |
| <input type="button" id="showoptions" name="showoptions" value="<%= loc 'Options' %>" /> |
| </form> |
| <div id="dropdownoptions" style="display: none"> |
| <form id="optionsform"> |
| <div class="controlgroup-vertical"> |
| <label for="cutoff">cut-off</label> |
| <input id="cutoff" type="text" name="cutoff" size="10" value="<%= $cutoff %>" title="Only consider the most frequent x word forms."> |
| <label for="dedupe">dedupe</label> |
| <input id="dedupe" type="checkbox" name="dedupe" value="1" <%= ($dedupe ? "checked" : "") %> title="radically filter out any near-duplicates"> |
| % if($mergedEnd > 0) { |
| <label for="sbf">backw.</label> |
| <input id="sbf" type="checkbox" name="sbf" value="1" <%= ($searchBaseVocabFirst ? "checked" : "") %> title="If checkecked base vocabulary will be searched first. Otherwise merged vocabulray will be searched first."> |
| % } |
| <label for="neighbours">max. neighbours:</label> |
| <input id="neighbours" size="4" name="n" value="<%= $no_nbs %>"> |
| <label for="no_iterations">max. iterations</label> |
| <input id="no_iterations" name="N" size="4" value="<%= $no_iterations %>"> |
| <!-- <label for="dosom">SOM</label> |
| <input id="dosom" type="checkbox" name="som" value="1" <%= ($show_som ? "checked" : "") %>> --> |
| % if($collocators) { |
| <label for="sortby">window/sort</label> |
| <select id="sortby" name="sort"> |
| <option value="0" <%= ($sort!=1 && $sort!=2? "selected":"") %>>auto focus</option> |
| <!-- <option value="1" <%= ($sort==1? "selected":"") %>>any single position</option> |
| <option value="2" <%= ($sort==2? "selected":"") %>>whole window</option> --> |
| </select> |
| % } |
| <input type="button" value="→ KorAP" onclick="queryKorAP();" title="query word with KorAP"/> |
| <input id="show-details" type="checkbox" name="show-details" value="1" > |
| <label for="show-details"> |
| Show details |
| </label> |
| </div> |
| </form> |
| </div> |
| </div> |
| </div> |
| %# Operands that are not in the vocabulary are dropped from the query |
| %# vector, which for an expression like "König - Mannn + Frau" would |
| %# otherwise silently answer a different question. |
| % if($unknown ne '' && $have_results) { |
| <div id="unknownwords"><%= loc 'not_in_vocabulary' %> <span class="mono"><%= $unknown %></span></div> |
| % } |
| <div id="topwrapper"> |
| <div style="visibility: hidden;" id="tabs"> |
| <ul> |
| % if (defined $word && $word ne '') { |
| % if($mergedEnd && $distantWords) { |
| <li><a href="#tabs-0" title="Cos offsets of the words furthest away from their position in the reference corpus."">Offsets</a></li> |
| % } |
| <li><a href="#tabs-1"><%= loc 'paradigmatic_tsne' %></a></li> |
| <li><a href="#tabs-2"><%= loc 'paradigmatic_som' %></a></li> |
| %# No syntagmatic tab for a model without a .net file, i.e. without |
| %# the output weights the predictive collocators are read from. |
| % if($collocators) { |
| <li><a href="#tabs-3"><%= loc 'syntagmatic' %></a></li> |
| % } |
| % } |
| <li><a href="#tabs-4">Info</a></li> |
| </ul> |
| % if($mergedEnd && $distantWords) { |
| <div id="tabs-0" style="display: flex; padding: 5px; flex-flow: row wrap;"> |
| <div id="vocabdist" style="width: 230px; margin-bottom: 15px;"> |
| <table class="display compact nowrap" id="vocabdistt"> |
| <thead> |
| <tr> |
| <th align="right">#</th><th id="cosD" align="right">D<sub>cos</sub></th><th align="left">word</th> |
| </tr> |
| </thead> |
| <tbody> |
| <tr> |
| <td align="right"> |
| </td> |
| <td align="right"> |
| </td> |
| <td></td> |
| </tr> |
| </tbody> |
| </table> |
| </div> |
| </div> |
| % } |
| <div id="tabs-1" style="display: flex; padding: 5px; flex-flow: row wrap;"> |
| % if($have_results) { |
| <div id="wrapper"> |
| <div id="first" style="width: 230px; margin-bottom: 15px;"> |
| <table class="display compact nowrap" id="firsttable"> |
| <thead> |
| <tr> |
| <th align="right">#</th><th align="right">S<sub>cos</sub></th><th align="left">similars by w2v</th> |
| </tr> |
| </thead> |
| <tbody> |
| <tr> |
| <td align="right"> |
| </td> |
| <td align="right"> |
| </td> |
| <td></td> |
| </tr> |
| </tbody> |
| </table> |
| </div> |
| % if(0 && $haveSProfiles) { |
| <div id="sprofilesdiv" style="width: 200px; padding-right: 10px;"> |
| <table class="display compact nowrap" id="sprofiles"> |
| <thead> |
| <tr> |
| <th align="right">cos</th><th align="left">similars by coll. profile</th> |
| </tr> |
| </thead> |
| <tbody> |
| <tr> |
| <td align="right"> |
| </td> |
| <td></td> |
| </tr> |
| </tbody> |
| </table> |
| </div> |
| %} |
| <div id="second"> |
| <div id="embed"> |
| </div> |
| <div id="cost"> |
| </div> |
| </div> |
| </div> |
| % } elsif($word !~ /^\s*$/) { |
| <div id="wrapper"> |
| <script> |
| $( function() { |
| $( "<%== loc 'notfounddialog' %>").dialog({ |
| autoOpen: true, |
| modal: true, |
| draggable: false, |
| height: "auto", |
| width: "auto", |
| resizable: false, |
| buttons: { |
| "OK": function() { |
| $( this ).dialog( "close" ); |
| } /* , |
| "Apply": function() { |
| window.open($(location).attr('pathname')+'?'+$('form').serialize(), "_self"); |
| } */ |
| } |
| }); |
| }); |
| </script> |
| <div id="not-found-dialog_de" style="display: none" title="Nicht gefunden"> |
| <p>FEHLER: Konnte "<%= $word %>" nicht finden.</p> |
| <p>Wenn Sie der Meinung sind, dass es Vokabluar enthalten sein sollte, können Sie versuchen den Cut-Off-Parameter in den Optionen zu erhöhen.</p> |
| </div> |
| <div id="not-found-dialog_en" style="display: none" title="Not found"> |
| <p>ERROR: "<%= $word %>" not found in vocabluary.</p> |
| <p>If you are sure you have spelled the word as intended, you can try to increase the cutoff parameter in the options menu.</p> |
| </div> |
| </div> |
| % } |
| </div> |
| <div id="tabs-2" style="display: flex; padding: 5px; flex-flow: row wrap;"> |
| % if(defined $word && $word ne "") { |
| <div id="som2" style="width: 800;"> |
| <div id="sominfo1"><span id="somcolor1"> </span> <span id="somword1"> </span> <span id="somcolor2"> </span> <span id="somword2"> </span> <span id="somcolor3"> </span></div> |
| <div id="sominfo" style="text-align: right">SOM iteration <span id="iterations">0</span></div> |
| </div> |
| % } |
| </div> |
| % if($collocators) { |
| <div id="tabs-3" style="display: flex; padding:5px; flex-flow: row wrap;"> |
| <div style="margin-right: 20px; margin-bottom: 10px;" id="secondt"> |
| <table class="display compact nowrap" id="secondtable"> |
| <thead> |
| <tr> |
| % if($collocators) { |
| <th>#</th> |
| <th align="center" title="Activation of the respective collocator in the columns around the target normalized by its maximum (red). Columns selected by the auto-focus funtion (which window of all possible column-combinations maximizes ⊥(a/c)?) are marked with +. Click on the column postions to lauch a KorAP query with target word and collocator in the respective position.">w'</th> |
| <th align="right" title="Maximum activation of the collocator anywhere in the output layer.">max(a)</th> |
| <th title="Average raw activation of the collocator in the columns selected by auto-focus." align="right">⟨a⟩</th> |
| <th title="Sum of activations over the selected colunns normalized by the total activation sum of the selected columns." align="right">Σa/Σw'</th> |
| <th title="Co-norm of the column-normalized activations over the colunns selected by the auto-focus." align="right">⊥(a/c)</th> |
| <th title="Sum of the activations over the whole window normalized by the total window sum (no auto-focus)." align="right">Σa/Σw</th> |
| <th align="left"><%= loc 'collocate_w2v' %></th> |
| % } |
| </tr> |
| </thead> |
| <tbody> |
| <tr> |
| <td align="right"> |
| </td> |
| <td align="right"> |
| </td> |
| <td align="right"> |
| </td> |
| <td align="right"> |
| </td> |
| <td align="right"> |
| </td> |
| <td align="right"> |
| </td> |
| </tr> |
| </tbody> |
| </table> |
| </div> |
| % if(!$single_word) { |
| <div id="ccd" class="notice"><%= loc 'ca_single_word_only' %></div> |
| % } else { |
| <div id="ccd" style=""> |
| <table class="display compact nowrap" id="classicoloctable"> |
| <thead> |
| % if($collocators) { |
| <tr> |
| <th id="ll_tt">LL</th> |
| <th id="mi_tt">MI</th> |
| <th id="md_tt">MI²</th> |
| <th id="lfmd_tt">MI³</th> |
| <th id="npmi_tt">nPMI</th> |
| <th id="dice_tt">dice</th> |
| <th id="logdice_tt">LD</th> |
| <th id="logdiceaf_tt">LDaf</th> |
| <th id="delta_tt" title="Delta to log-Dice score in reference corpus. ⚠: If the collocate is not within the top 200 of the reference corpus, a reference value of min(lD)-0.1 is assumed.">Δ</th> |
| <th id="af_win" title="Positions around the target word that are selected by the auto-focus function are marked with ◾. Positions at which the collocate appears at least once are marked with ◽. LL, MI, MI², MI³ and nPMI are computed over all positions marked here, the attested ones, rather than over the whole context window, so that a collocate occurring in few positions is rated higher than one spread over many. LD does not depend on any window; LDaf is the auto-focus score of the ◾ positions and is not on the same scale as LD."><span class="regular"><%= loc 'af_window' %></span></th> |
| <th title="PMI³ restricted to left neighbour">l-PMI³</th> |
| <th title="PMI³ restricted to right neighbour">r-PMI³</th> |
| <th title="nPMI restricted to left neighbour">l-nPMI</th> |
| <th title="nPMI restricted to right neighbour">r-nPMI</th> |
| <th id="rawfreq_tt" title="raw frequency sum of collocation within window">raw</th> |
| <th><%= loc 'collocate_ca' %></th> |
| </tr> |
| % } |
| </thead> |
| <tbody> |
| <tr> |
| <td align="right"></td> |
| <td align="right"></td> |
| <td align="right"></td> |
| <td align="right"></td> |
| <td align="right"></td> |
| <td align="right"></td> |
| <td align="right"></td> |
| <td align="right"></td> |
| <td align="right"></td> |
| <td align="left"> </td> |
| <td align="right"></td> |
| <td align="right"></td> |
| <td align="right"></td> |
| <td align="right"></td> |
| <td align="left"> </td> |
| </tr> |
| </tbody> |
| </table> |
| </div> |
| % } |
| <!-- |
| <div style="clear:both" ></div> |
| <div style="float: right; overflow: hidden" id="extra"><button onClick="showCollocatorSOM()"> </button></div> |
| --> |
| </div> |
| % } |
| <div id="tabs-4" style="display: flex; padding:5px; flex-flow: row wrap;"> |
| <div id="info"> |
| <h3><%== loc 'about' %></h3> |
| %= include(loc('abouttext')) |
| % if($training_args && (@$lists)[0]) { |
| <h3><%= loc 'training_parameters' %></h3> |
| % if($training_args =~ /-type\s*3/) { |
| <p>Calculations are based on a word embedding model trained with an extension of <a href="https://github.com/wlin12/wang2vec/">wang2vec</a> using the following parameters:</p> |
| <div class="mono"><%= $training_args %></div> |
| % } else { |
| <p>Calculations are based on a word embedding model trained with <a href="https://code.google.com/p/word2vec/">word2vec</a> using the following parameters:</p> |
| <div class="mono"><%= $training_args %></div> |
| % } |
| % } |
| <h3>Source Code</h3> |
| <ul> |
| <li><a href="https://github.com/kupietz/dereko2vec">dereko2vec</a></li> |
| <li><a href="https://github.com/kupietz/collocatordb">collocatordb</a></li> |
| <li><a href="https://korap.ids-mannheim.de/gerrit/plugins/gitiles/ids-kl/derekovecs">derekovecs</a></li> |
| </ul> |
| <h3><%== loc 'references' %></h3> |
| %= include 'references' |
| </div> |
| </div> |
| </div> <!-- tabs --> |
| </div> <!-- topwrapper --> |
| <div style="clear: both;"></div> |
| </div> |
| <div class="footer"> |
| <span class="footertext"> |
| %= include(loc('footer')) |
| </span> |
| </div> |
| </body> |
| </html> |