Crfと素性テンプレート
View more presentations from uchumik
TokyoNLP で話す内容を貼っておきます。
きっと後で訂正が入るんだろうな。。
日々の日記や自然言語処理関連の話題について書いていこうかなと思います。
# include "rtcrflearn.hpp"
# include "rtcrftagger.hpp"
using namespace RtCrf;
using namespace std;
int main(int argc, char **argv)
{
Crflearn learner(*(argv+1),*(argv+2),1000000);
learner.init();
learner.learn(5,0);
learner.save("test.model");
Crftagger tagger(*(argv+1),1000000);
tagger.read("test.model");
tagger.tagging(*(argv+2));
return 0;
}
# ./test template ../data/train.txt
labels: 2
bound: 3
ufeatures: 17
bfeatures: 1
instance: 1
uparameters: 34
bparameters: 8
model parameters: 42
epoch: 0 err:0.500000(2/4)
epoch: 1 err:0.000000(0/4)
epoch: 2 err:0.000000(0/4)
epoch: 3 err:0.000000(0/4)
epoch: 4 err:0.000000(0/4)
a b c 1.0 -1.0 L_A L_A
d e f -1.0 1.0 L_B L_B
p e f -0.5 0.1 L_B L_B
g b c 0.3 -0.1 L_A L_A
(x_1+x_2)*x_3*max(max(...),min(...),x_4,log(x_5),...)
# Unigram
U00:%x[-2,0]
U01:%x[-1,0]_%x[0,0]
U02:%x[-1,0]_%x[0,0]_%x[1,0]
U03:1.0-%r[0,3]
U04:%r[0,3]+%r[0,4]
U05:%r[0,3]-%r[0,4]
U06:%r[0,3]*-%r[0,4]
U07:%r[0,3]/-%r[0,4]
U08:%r[0,3]%-%r[0,4]
U09:SUM(%r[.,3])
U10:PROD(%r[.,4])
U11:MIN(%r[0,3],%r[0,4])
U12:MAX(%r[0,3],%r[0,4])
U13:MIN(%r[.,3])
U14:MAX(%r[.,4])
U15:MIN(%r[0,.])
U16:MAX(%r[0,.])
U17:MIN(MAX(%r[0,.],%r[.,3]),%r[.,4])
U18:%r[0,3]/MAX(%r[.,3])
U19:%r[0,3]/SUM(%r[.,3])
U20:MIN(%r[0,3]+%r[0,4],%r[0,3])
# Bigram
B
a b c 1.0 -1.0 L_A
d e f -1.0 1.0 L_B
p e f -0.5 0.1 L_B
g b c 0.3 -0.1 L_A
U00:_B-2:1 U01:_B-1_a:1 U02:_B-1_a_d:1 U05:2 U06:1 U07:1 U09:-0.2 U10:0.01 U11:-1 U12:1 U13:-1 U14:1 U15:-1 U16:1 U17:-1 U18:1 U19:-5 B:1
U00:_B-1:1 U01:a_d:1 U02:a_d_p:1 U03:2 U05:-2 U06:1 U07:1 U09:-0.2 U10:0.01 U11:-1 U12:1 U13:-1 U14:1 U15:-1 U16:1 U17:-1 U18:-1 U19:5 U20:-1 B:1
U00:a:1 U01:d_p:1 U02:d_p_g:1 U03:1.5 U04:-0.4 U05:-0.6 U06:0.05 U07:5 U08:-0.1 U09:-0.2 U10:0.01 U11:-0.5 U12:0.1 U13:-1 U14:1 U15:-0.5 U16:0.1 U17:-1 U18:-0.5 U19:2.5 U20:-0.5 B:1
U00:d:1 U01:p_g:1 U02:p_g__E+1:1 U03:0.7 U04:0.2 U05:0.4 U06:0.03 U07:3 U08:0.1 U09:-0.2 U10:0.01 U11:-0.1 U12:0.3 U13:-1 U14:1 U15:-0.1 U16:0.3 U17:-1 U18:0.3 U19:-1.5 U20:0.2 B:1
for (k = 0; k < L; ++k)
alpha_{i,j} = logsumexp(alpha_{i,j}, alpha_{i-1,k}+cost)
# Bigram
B00:%x[-2,0,0]
B01:%x[-1,0,0]
B02:%x[0,0,0]
B03:%x[1,0,0]
B04:%x[2,0,0]
B05:%x[-1,0,0]/%x[0,0,0]
B06:%x[0,0,0]/%x[1,0,0]
B10:%x[-2,1,0]
B11:%x[-1,1,0]
B12:%x[0,1,0]
B13:%x[1,1,0]
B14:%x[2,1,0]
B15:%x[-2,1,0]/%x[-1,1,0]
B16:%x[-1,1,0]/%x[0,1,0]
B17:%x[0,1,0]/%x[1,1,0]
B18:%x[1,1,0]/%x[2,1,0]
B20:%x[-2,1,0]/%x[-1,1,0]/%x[0,1,0]
B21:%x[-1,1,0]/%x[0,1,0]/%x[1,1,0]
B22:%x[0,1,0]/%x[1,1,0]/%x[2,1,0]
B
time ./src/learner --learner=cnf -t src/template -c data/conll2000/train.txt -s bigramtest.save --l2 --pool=1000000 --block=24 --lambda=2 --iter=1
labels: 22
features: 338552
bound: 3
ufeatures: 0
bfeatures: 76329
instance: 8936
gate functions: 20
parameters of gate functions: 31
uparameters: 0
bparameters: 40301712
model parameters: 40301743
epoch: 0 err:0.069070(14624/211727)
./src/learner --learner=cnf -t src/template -c data/conll2000/train.txt -s 1323.90s user 41.84s system 96% cpu 23:40.72 total
U00:%x[-2,0,0]
U01:%x[-1,0,0]
U02:%x[0,0,0]
U03:%x[1,0,0]
U04:%x[2,0,0]
U05:%x[-1,0,0]/%x[0,0,0]
U06:%x[0,0,0]/%x[1,0,0]
U10:%x[-2,1,0]
U11:%x[-1,1,0]
U12:%x[0,1,0]
U13:%x[1,1,0]
U14:%x[2,1,0]
U15:%x[-2,1,0]/%x[-1,1,0]
U16:%x[-1,1,0]/%x[0,1,0]
U17:%x[0,1,0]/%x[1,1,0]
U18:%x[1,1,0]/%x[2,1,0]
U20:%x[-2,1,0]/%x[-1,1,0]/%x[0,1,0]
U21:%x[-1,1,0]/%x[0,1,0]/%x[1,1,0]
U22:%x[0,1,0]/%x[1,1,0]/%x[2,1,0]
# Bigram
B
time ./src/learner --learner=cnf -t src/template -c data/conll2000/train.txt -s bigramtest.save --l2 --pool=1000000 --block=24 --lambda=2 --iter=1
labels: 22
features: 338552
bound: 3
ufeatures: 76328
bfeatures: 1
instance: 8936
gate functions: 20
parameters of gate functions: 31
uparameters: 1679216
bparameters: 528
model parameters: 1679775
epoch: 0 err:0.057314(12135/211727)
./src/learner --learner=cnf -t src/template -c data/conll2000/train.txt -s 67.23s user 5.61s system 92% cpu 1:18.67 total
# Unigram Segment
S:%s[15,0]
# Bigram Segment
T
# Unigram
U00:%x[-2,0,0]
U01:%x[-1,0,0]
U02:%x[0,0,0]
U03:%x[1,0,0]
U04:%x[2,0,0]
U05:%x[-1,0,0]/%x[0,0,0]
U06:%x[0,0,0]/%x[1,0,0]
U10:%x[-2,1,0]
U11:%x[-1,1,0]
U12:%x[0,1,0]
U13:%x[1,1,0]
U14:%x[2,1,0]
U15:%x[-2,1,0]/%x[-1,1,0]
U16:%x[-1,1,0]/%x[0,1,0]
U17:%x[0,1,0]/%x[1,1,0]
U18:%x[1,1,0]/%x[2,1,0]
U20:%x[-2,1,0]/%x[-1,1,0]/%x[0,1,0]
U21:%x[-1,1,0]/%x[0,1,0]/%x[1,1,0]
U22:%x[0,1,0]/%x[1,1,0]/%x[2,1,0]
# Bigram
B
# Unigram Segment
S:%s[15,0]
# Bigram Segment
T
# make
# ./src/semicnflearn src/semitemplate data/conll2000/train.txt semicnf.save
# ./src/semicnftagger src/semitemplate semicnf.save data/conll2000/test.txt
# Unigram Segment
S01:%s[15,0,0]
S02:%s[15,1,0]
# Bigram Segment
T
# Unigram
U00:%x[-2,0,0]
U01:%x[-1,0,0]
U02:%x[0,0,0]
U03:%x[1,0,0]
U04:%x[2,0,0]
U05:%x[-1,0,0]/%x[0,0,0]
U06:%x[0,0,0]/%x[1,0,0]
U10:%x[-2,1,0]
U11:%x[-1,1,0]
U12:%x[0,1,0]
U13:%x[1,1,0]
U14:%x[2,1,0]
U15:%x[-2,1,0]/%x[-1,1,0]
U16:%x[-1,1,0]/%x[0,1,0]
U17:%x[0,1,0]/%x[1,1,0]
U18:%x[1,1,0]/%x[2,1,0]
U20:%x[-2,1,0]/%x[-1,1,0]/%x[0,1,0]
U21:%x[-1,1,0]/%x[0,1,0]/%x[1,1,0]
U22:%x[0,1,0]/%x[1,1,0]/%x[2,1,0]
# Bigram
B
#!/opt/local/bin/perl
# user id | item id | rating | timestamp
use warnings;
use strict;
my %dic;
my $file = "../ml-data/u.item";
open (F, $file) or die;
while ()
{
chomp;
my @t = split(/\|/);
unless (defined $dic{$t[0]})
{
$dic{$t[0]} = $t[1];
}
}
close (F);
while (<>)
{
chomp;
my @tokens = split(/\t/);
print $dic{$tokens[1]}, "\t$tokens[0]\t$tokens[2]\n";
}
'Til There Was You (1997) 152 4
'Til There Was You (1997) 178 3
'Til There Was You (1997) 223 1
'Til There Was You (1997) 299 2
'Til There Was You (1997) 342 1
'Til There Was You (1997) 416 3
'Til There Was You (1997) 530 2
'Til There Was You (1997) 532 3
'Til There Was You (1997) 782 2
...
Snow White and the Seven Dwarfs (1937)
[Instance]
Snow White and the Seven Dwarfs (1937):0.999801
Cinderella (1950):5.19753e-07
Fantasia (1940):5.06323e-07
Dumbo (1941):4.17243e-07
Beauty and the Beast (1991):3.94127e-07
Aladdin (1992):3.88318e-07
Mary Poppins (1964):3.71035e-07
Jungle Book, The (1994):3.67852e-07
Sword in the Stone, The (1963):3.67118e-07
Lion King, The (1994):3.64219e-07
Pinocchio (1940):3.60772e-07
Robin Hood: Prince of Thieves (1991):3.5665e-07
Aristocats, The (1970):3.51889e-07
Fox and the Hound, The (1981):3.34974e-07
Star Trek VI: The Undiscovered Country (1991):3.24104e-07
Alice in Wonderland (1951):3.19468e-07
Homeward Bound: The Incredible Journey (1993):2.97177e-07
Sound of Music, The (1965):2.97109e-07
Walk in the Sun, A (1945):2.96338e-07
Very Natural Thing, A (1974):2.96338e-07
Bedknobs and Broomsticks (1971):2.93205e-07
Star Trek IV: The Voyage Home (1986):2.90837e-07
Gone with the Wind (1939):2.89412e-07
Nightmare Before Christmas, The (1993):2.85079e-07
Star Trek: Generations (1994):2.8431e-07
Winnie the Pooh and the Blustery Day (1968):2.82242e-07
101 Dalmatians (1996):2.81954e-07
Swiss Family Robinson (1960):2.6257e-07
Aladdin and the King of Thieves (1996):2.54459e-07
Grease (1978):2.53118e-07
African Queen, The (1951):2.52688e-07
Ghost (1990):2.52399e-07
Hunt for Red October, The (1990):2.4501e-07
Miracle on 34th Street (1994):2.42372e-07
Home Alone (1990):2.4004e-07
Top Gun (1986):2.35881e-07
Abyss, The (1989):2.35302e-07
Little Women (1994):2.33329e-07
Old Yeller (1957):2.32874e-07
Pocahontas (1995):2.31783e-07
Mrs. Doubtfire (1993):2.31762e-07
Jumanji (1995):2.31231e-07
Pete's Dragon (1977):2.29198e-07
Christmas Carol, A (1938):2.28481e-07
New York Cop (1996):2.28269e-07
Maltese Falcon, The (1941):2.26072e-07
Goofy Movie, A (1995):2.25177e-07
James and the Giant Peach (1996):2.23962e-07
Star Trek: The Motion Picture (1979):2.23878e-07
Walk in the Sun, A (1945)
Star Trek: Generations (1994)
[Instance]
Snow White and the Seven Dwarfs (1937):0.9998
Cinderella (1950):2.96839e-07
Fantasia (1940):2.81548e-07
Dumbo (1941):2.54658e-07
Aladdin (1992):2.4008e-07
Lion King, The (1994):2.30137e-07
Beauty and the Beast (1991):2.10186e-07
I Don't Want to Talk About It (De eso no se habla) (1993):2.01313e-07
Jurassic Park (1993):1.96396e-07
Coldblooded (1995):1.95535e-07
Robin Hood: Prince of Thieves (1991):1.74025e-07
Fox and the Hound, The (1981):1.71627e-07
African Queen, The (1951):1.69855e-07
Sound of Music, The (1965):1.69781e-07
Mary Poppins (1964):1.64425e-07
Wizard of Oz, The (1939):1.6242e-07
Swan Princess, The (1994):1.56013e-07
Homeward Bound: The Incredible Journey (1993):1.53814e-07
Mamma Roma (1962):1.52174e-07
Good Man in Africa, A (1994):1.50378e-07
Some Like It Hot (1959):1.46221e-07
Citizen Kane (1941):1.42743e-07
Anne Frank Remembered (1995):1.40035e-07
Dances with Wolves (1990):1.36046e-07
It's a Wonderful Life (1946):1.3514e-07
Hunt for Red October, The (1990):1.3501e-07
Young Frankenstein (1974):1.34608e-07
Rebecca (1940):1.32806e-07
Winnie the Pooh and the Blustery Day (1968):1.30989e-07
It Happened One Night (1934):1.30592e-07
M (1931):1.30189e-07
Maltese Falcon, The (1941):1.29532e-07
Lassie (1994):1.28929e-07
Foreign Correspondent (1940):1.28287e-07
Grease (1978):1.2729e-07
Alice in Wonderland (1951):1.23873e-07
Quiz Show (1994):1.22688e-07
Singin' in the Rain (1952):1.22222e-07
Sword in the Stone, The (1963):1.18912e-07
Lawrence of Arabia (1962):1.15236e-07
To Kill a Mockingbird (1962):1.14732e-07
Swiss Family Robinson (1960):1.14698e-07
Drop Dead Fred (1991):1.13925e-07
Vertigo (1958):1.13817e-07
Top Gun (1986):1.12622e-07
Bonnie and Clyde (1967):1.12262e-07
Blue Angel, The (Blaue Engel, Der) (1930):1.10351e-07
Annie Hall (1977):1.09473e-07
Turning, The (1992):1.07052e-07
#!/usr/local/bin/perl
use warnings;
use strict;
my $L = 5;
my @dataset;
my %wordset;
my $words = 0;
while(<>)
{
chomp;
my $line = $_;
my @sequence = split(//,$line);
push @dataset,\@sequence;
}
# init words and probs
$words = &setwords(\%wordset, \@dataset);
&initwordprobs($words,\%wordset);
my $datasize = @dataset;
print "dataset\n";
foreach my $ar (@dataset)
{
print "@$ar\n";
}
# init motif set
my @pos;
foreach my $ar (@dataset)
{
my $p = int rand(@$ar-$L);
push @pos, $p;
#print $p,"\n";
}
# output first
print "init\n";
for (my $i = 0; $i < @dataset; $i++)
{
my $ar = $dataset[$i];
my $p = $pos[$i];
for (my $j = $p; $j < $p+$L; $j++)
{
print "$ar->[$j] ";
}
print "\n";
}
# iterration
for (my $iter = 0; $iter < 50; $iter++)
{
#print "epoch: $iter\n";
my $id = int rand($datasize);
my %wordsfreq;
# wordset * position
# +---+-------------+
# | |a|b|c|d|e|f|g|
# +---+-------------+
# | 0 | |
# | 1 | |
# | 2 | frequency |
# | . | |
# | L | |
# +---+-------------+
&initwordsfreq(\%wordset,$L,\%wordsfreq);
# count word frequency
&countwordfreq($id,\@dataset,\%wordsfreq,\@pos);
# pick up new motif i
my $ar = $dataset[$id];
$pos[$id] = &pickup($id,$ar,$L,\%wordsfreq, \%wordset);
}
# output
print "result\n";
for (my $i = 0; $i < @dataset; $i++)
{
my $ar = $dataset[$i];
my $p = $pos[$i];
for (my $j = $p; $j < $p+$L; $j++)
{
print "$ar->[$j] ";
}
print "\n";
}
# subroutine
sub pickup
{
my $id = shift;
my $ar = shift;
my $L = shift;
my $wordsfreqhr = shift;
my $wordsethr = shift;
my @scores;
my $total = 0;
for (my $p = 0; $p <= @$ar-$L; $p++)
{
my $score = 1;
my $index = 0;
for (my $j = $p; $j < $p+$L; $j++)
{
my $w = $ar->[$j];
$score *= $wordsfreqhr->{$index}->{$w}/$wordsethr->{$w};
$index++;
}
push @scores, $score;
$total += $score;
}
return int rand(@scores) if $total == 0;
for(my $i = 0; $i < @scores; $i++)
{
$scores[$i] /= $total;
}
#print "@scores\n";
my $flg = 1;
while($flg)
{
my $npos = int rand(@scores);
my $s = rand(1);
if ($s <= $scores[$npos])
{
$flg = 0;
return $npos;
}
}
}
sub countwordfreq
{
my $id = shift;
my $datasetar = shift;
my $wordsfreqhr = shift;
my $posar = shift;
for (my $i = 0; $i < @$datasetar; $i++)
{
next if ($i == $id);
my $ar = $datasetar->[$i];
for (my $j = $posar->[$i]; $j < $L+$posar->[$i]; $j++)
{
my $p = $j-$posar->[$i];
my $w = $ar->[$j];
$wordsfreqhr->{$p}->{$w}++;
}
}
for (my $l = 0; $l < $L; $l++)
{
foreach my $k (keys %{$wordsfreqhr->{$l}})
{
$wordsfreqhr->{$l}->{$k} /= (@$datasetar-1);
}
}
}
sub initwordsfreq
{
my $wordsethr = shift;
my $L = shift;
my $wordfhr = shift;
for (my $i = 0; $i < $L; $i++)
{
my %freqonpos;
foreach my $k (keys %$wordsethr)
{
$freqonpos{$k} = 0;
}
$wordfhr->{$i} = \%freqonpos;
}
}
sub initwordprobs
{
my $words = shift;
my $wordsethr = shift;
foreach my $k (keys %$wordsethr)
{
$wordsethr->{$k} = 1/$words;
}
}
sub setwords
{
my $wordshr = shift;
my $dataar = shift;
my $words = 0;
foreach my $ar (@$dataar)
{
foreach my $w (@$ar)
{
unless (defined $wordshr->{$w})
{
$wordshr->{$w} = defined;
$words++;
}
}
}
return $words;
}
c a t c g a a a a a a
a a a a c a t c g a a
a a c a t c g a a a a
a a a a a a c a t c g
t c a g a g a t t c g t a g
t t c a t t c g g g c g t
c g a t t c g c g a c t c
t g a a t t c g g a a
# perl gibbs.pl < sample1.txt
dataset
c a t c g a a a a a a
a a a a c a t c g a a
a a c a t c g a a a a
a a a a a a c a t c g
init
g a a a a
a a a c a
a t c g a
a a a a a
result
c a t c g
c a t c g
c a t c g
c a t c g
# perl gibbs.pl < sample2.txt
dataset
t c a g a g a t t c g t a g
t t c a t t c g g g c g t
c g a t t c g c g a c t c
t g a a t t c g g a a
init
t c a g a
c a t t c
t c g c g
a a t t c
result
a t t c g
a t t c g
a t t c g
a t t c g