Support non-word-tokens (fixes #5) Change-Id: I6867745afd7c0fb865722bcd62a0724aaa9a6ccb

commit: ed9baf0cd3d987305c1db2a420af1a27c56b4a08 [log] [tgz]
author: Akron <nils@diewald-online.de> Tue Jan 22 17:03:25 2019 +0100
committer: Akron <nils@diewald-online.de> Tue Jan 22 17:03:25 2019 +0100
tree: 9246a824f52fa49465cd95a199214b60209aa765
parent: 6eff23bb1e216ca4c630ec935bcf0c3c4b44f11b [diff] [blame]
diff --git a/t/tokenization.t b/t/tokenization.t
new file mode 100644
index 0000000..4c4ddaa
--- /dev/null
+++ b/t/tokenization.t

@@ -0,0 +1,65 @@
+#!/usr/bin/env perl
+use strict;
+use warnings;
+use utf8;
+use Test::More;
+use JSON::XS;
+
+use File::Basename 'dirname';
+use File::Spec::Functions 'catdir';
+
+sub _t2h {
+  my $string = shift;
+  $string =~ s/^\[\(\d+?-\d+?\)(.+?)\]$/$1/;
+  my %hash = ();
+  foreach (split(qr!\|!, $string)) {
+    $hash{$_} = 1;
+  };
+  return \%hash;
+};
+
+use_ok('KorAP::XML::Krill');
+
+my $path = catdir(dirname(__FILE__), 'corpus/WPD/00001');
+ok(my $doc = KorAP::XML::Krill->new( path => $path ), 'Load Korap::Document');
+like($doc->path, qr!\Q$path\E/$!, 'Path');
+ok($doc->parse, 'Parse document');
+is($doc->text_sigle, 'WPD/AAA/00001', 'ID');
+
+
+# Get tokens
+use_ok('KorAP::XML::Tokenizer');
+
+# Get tokenization
+ok(my $tokens = KorAP::XML::Tokenizer->new(
+  path => $doc->path,
+  doc => $doc,
+  foundry => 'OpenNLP',
+  layer => 'Tokens',
+  name => 'tokens'
+), 'New Tokenizer');
+ok($tokens->parse, 'Parse');
+
+like($tokens->stream->pos(12)->to_string, qr/s:Vokal/);
+like($tokens->stream->pos(13)->to_string, qr/s:Der/);
+
+
+# Get tokenization with non word tokens
+ok($tokens = KorAP::XML::Tokenizer->new(
+  path => $doc->path,
+  doc => $doc,
+  foundry => 'OpenNLP',
+  layer => 'Tokens',
+  name => 'tokens',
+  non_word_tokens => 1
+), 'New Tokenizer');
+ok($tokens->parse, 'Parse');
+
+like($tokens->stream->pos(12)->to_string, qr/s:Vokal/);
+like($tokens->stream->pos(13)->to_string, qr/s:\./);
+like($tokens->stream->pos(14)->to_string, qr/s:Der/);
+
+
+done_testing;
+
+__END__
commit	ed9baf0cd3d987305c1db2a420af1a27c56b4a08	[log] [tgz]
author	Akron <nils@diewald-online.de>	Tue Jan 22 17:03:25 2019 +0100
committer	Akron <nils@diewald-online.de>	Tue Jan 22 17:03:25 2019 +0100
tree	9246a824f52fa49465cd95a199214b60209aa765
parent	6eff23bb1e216ca4c630ec935bcf0c3c4b44f11b [diff] [blame]