[ Skip to the content ]

Institute of Formal and Applied Linguistics Wiki


[ Back to the navigation ]

Differences

This shows you the differences between two versions of the page.

Link to this comparison view

Next revision Both sides next revision
user:zeman:treebanks:hi [2011/12/06 16:24]
zeman vytvořeno
user:zeman:treebanks:hi [2011/12/06 16:32]
zeman Sample training Shakti.
Line 87: Line 87:
 ==== Sample ==== ==== Sample ====
  
-The first sentence of the ICON 2010 training data (with fine-grained syntactic tags) in the Shakti format:+The first two sentences of the ICON 2010 training data (with fine-grained syntactic tags) in the Shakti format:
  
-<code xml><document id=""> +<code xml><document docid="hi">
 <head> <head>
-<annotated-resource name="HyDT-Bangla" version="0.5" type="dep-interchunk-only" layers="morph,pos,chunk,dep-interchunk-only" language="ben" date-of-release="20100831">+<title>  </title>  
 +<author>  
 +<firstname>  </firstname>  
 +<middlename>    </middlename>  
 +<lastname></lastname>  
 +</author>  
 +<availability format="electronic" />  
 +<bibl>  
 +</bibl>  
 +<bytecount>8.0K</bytecount>  
 +<domain name="general" />  
 +<creation creationdate="19/06/2007" institutename="IIIT Hyderabad">  
 +<creatorname>  
 +<lastname>Dipti</lastname>  
 +<middlename>  
 +</middlename>  
 +<firstname>Sharma</firstname>  
 +</creatorname>  
 +</creation>  
 +<distributor>CLIA Consortia, DIT</distributor>  
 +<edition number="1.0" />  
 +<encodingdesc>  
 +<newencoding>Unicode(UTF-8)</newencoding>  
 +<originalencoding>UTF-8</originalencoding>  
 +</encodingdesc>  
 +<sentencemarker marker=".">Specify Marker</sentencemarker>  
 +<language name="hi" writingsystem="LTR" script="Devanagari" />  
 +<normalization normalized="no">  
 +<utilityname>xxx.exe</utilityname>  
 +</normalization>  
 +<projectdesc name="ILMT" />  
 +<pubaddress addresstype="web">  
 +</pubaddress>  
 +<pubdate>  
 +<dateofpublication></dateofpublication>  
 +</pubdate>  
 +<publicationstmt type="copyrightfree">  
 +</publicationstmt>  
 +<publisher>  
 +<name></name>  
 +<url>xxx.com</url>  
 +</publisher>  
 +<pubplace place="books" />  
 +<wordcount> </wordcount>  
 +<caption>xuvryavahAra se biParIM bipASA Pilma mahowsava se vApasa lOta gaI bipASA govA. </caption>  
 +</caption>  
 + 
 +<annotated-resource name="HyDT-Hindi" version="2.0" type="dep-words" layers="morph,pos,chunk,dep-word" language="hin" date-of-release="20100823">
     <annotation-standard>     <annotation-standard>
         <morph-standard name="Anncorra-morph" version="1.31" date="20080920" />         <morph-standard name="Anncorra-morph" version="1.31" date="20080920" />
         <pos-standard name="Anncorra-pos" version="" date="20061215" />         <pos-standard name="Anncorra-pos" version="" date="20061215" />
         <chunk-standard name="Anncorra-chunk" version="" date="20061215" />         <chunk-standard name="Anncorra-chunk" version="" date="20061215" />
 +        <intrachunk-dependency-standard name="Anncorra-intrachunk-dep" version="1.0" date="" dep-tagset-granularity="5" />
         <dependency-standard name="Anncorra-dep" version="2.0" date="" dep-tagset-granularity="6" />         <dependency-standard name="Anncorra-dep" version="2.0" date="" dep-tagset-granularity="6" />
     </annotation-standard>     </annotation-standard>
-</annotated-resource>  +</annotated-resource> 
-</head> +</head> 
 +<body> 
 +<tb number="1" segment="no" bullet="no"> 
 +<foreign language="select" writingsystem="LTR"></foreign> 
 +<text>
 <Sentence id="1"> <Sentence id="1">
-1 (( NP <fs af='Age,adv,,,,,,' head="Agei" drel=k7t:VGF name=NP+1 bAwa NN <fs af='bAwa,n,f,sg,3,d,0,0' drel='k1:ho' posn='10' name='bAwa' chunkId='NP' chunkType='head:NP'> 
-1.1 mudZira NN <fs af='mudZi,n,,sg,,o,era,era'> +2 galawa JJ <fs af='galawa,adj,any,any,,any,,' drel='k1s:ho' posn='20' name='galawa' chunkId='JJP' chunkType='head:JJP'> 
-1.2 Agei NST <fs af='Age,adv,,,,,,' name="Agei"> +3 ho VM <fs af='ho,v,any,any,any,,0,0' drel='vmod:hE' stype='declarative' posn='30' voicetype='active' name='ho' chunkId='VGF' chunkType='head:VGF'> 
- ))  +4 wo CC <fs af='wo,avy,,,,,,' posn='40' name='wo' chunkId='CCP' chunkType='head:CCP'
-2 (( NP <fs af='cA,n,,sg,,d,0,0' head="cA" drel=k1:VGF name=NP2+5 gussA NN <fs af='gussA,n,m,sg,3,d,0,0' drel='pof:AnA' posn='50' name='gussA' chunkId='NP2' chunkType='head:NP2'> 
-2.1 praWama QO <fs af='praWama,num,,,,,,'> +6 selebritija NN <fs af='selebritija,unk,,,,,0_ko,' drel='k4a:AnA' posn='60' vpos='vib_2_RP' name='selebritija' chunkId='NP3' chunkType='head:NP3'> 
-2.2 kApa NN <fs af='kApa,unk,,,,,,'> +7 ko PSP <fs af='ko,psp,,,,,,' posn='70' drel='lwg__psp:selebritija' chunkType='child:NP3' name='ko'> 
-2.3 cA NN <fs af='cA,n,,sg,,d,0,0' name="cA"+8 BI RP <fs af='BI,avy,,,,,,' posn='80' drel='lwg__rp:selebritija' chunkType='child:NP3' name='BI'> 
- ))  +9 AnA VM <fs af='A,v,any,any,any,d,nA,nA' drel='k1:hE' posn='90' name='AnA' chunkId='VGNN' chunkType='head:VGNN'> 
-3 (( VGF <fs af='As,v,,,5,,A_yA+Ce,A' head="ese" name=VGF+10 lAjamI JJ <fs af='lAjamI,adj,any,any,,,,' drel='pof:hE' posn='100' name='lAjamI' chunkId='JJP2' chunkType='head:JJP2'> 
-3.1 ese VM <fs af='As,v,,,7,,A,A' name="ese"+11 hE VM <fs af='hE,v,any,sg,3,,hE,hE' drel='ccof:wo' stype='declarative' posn='110' voicetype='active' name='hE' chunkId='VGF2' chunkType='head:VGF2'> 
-3.2 . SYM <fs af='.,punc,,,,,,'> +12 . SYM <fs af='.,punc,,,,,,' posn='120' drel='rsym:hE' chunkType='child:VGF2' name='.'> 
- )) +</Sentence> 
 + 
 + 
 +<Sentence id="2"> 
 +1 bqhaspawivAra NNP <fs af='bqhaspawivAra,n,m,sg,3,o,0_ko,0' drel='k7t:hue' posn='10' vpos='vib_2' name='bqhaspawivAra' chunkId='NP' chunkType='head:NP'> 
 +2 ko PSP <fs af='ko,psp,,,,,,' posn='20' drel='lwg__psp:bqhaspawivAra' chunkType='child:NP' name='ko'> 
 +3 jZI NNP <fs af='jI,n,m,sg,3,o,0_meM,0' drel='k7:hue' posn='30' vpos='vib_2' name='jZI' chunkId='NP2' chunkType='head:NP2'> 
 +4 meM PSP <fs af='meM,psp,,,,,,' posn='40' drel='lwg__psp:jZI' chunkType='child:NP2' name='meM'> 
 +5 SurU NN <fs af='SurU,n,m,sg,3,d,0,0' drel='pof:hue' posn='50' name='SurU' chunkId='NP3' chunkType='head:NP3'> 
 +6 hue VM <fs af='ho,v,m,sg,any,,eM,eM' drel='nmod__k1inv:mahowsava' posn='60' name='hue' chunkId='VGNF' chunkType='head:VGNF'
 +7 ��veM XC <fs af='��veM,n,m,sg,3,d,0,0' posn='70' drel='mod:mahowsava' chunkType='child:NP4' name='��veM'> 
 +8 aMwarrARtrIya XC <fs af='aMwarrARtrIya,n,m,sg,3,d,0,0' posn='80' drel='mod:mahowsava' chunkType='child:NP4' name='aMwarrARtrIya'> 
 +9 Pilma XC <fs af='Pilma,n,f,sg,3,d,0,0' posn='90' drel='mod:mahowsava' chunkType='child:NP4' name='Pilma'> 
 +10 mahowsava NNP <fs af='mahowsava,n,m,sg,,o,0_kA,0' drel='r6:raMga' posn='100' vpos='vib_5' name='mahowsava' chunkId='NP4' chunkType='head:NP4'> 
 +11 ke PSP <fs af='kA,psp,m,sg,,o,,' posn='110' drel='lwg__psp:mahowsava' chunkType='child:NP4' name='ke'> 
 +12 raMga NN <fs af='raMga,n,m,sg,3,o,0_meM,0' drel='k7:padZA' posn='120' vpos='vib_2' name='raMga' chunkId='NP5' chunkType='head:NP5'> 
 +13 meM PSP <fs af='meM,psp,,,,,,' posn='130' drel='lwg__psp:raMga' chunkType='child:NP5' name='meM2'> 
 +14 BaMga JJ <fs af='BaMga,adj,any,any,,any,,' drel='pof:padZA' posn='140' name='BaMga' chunkId='JJP' chunkType='head:JJP'> 
 +15 usa DEM <fs af='vaha,pn,any,sg,3,o,,' posn='150' drel='nmod__adj:samaya' chunkType='child:NP6' name='usa'> 
 +16 samaya NN <fs af='samaya,n,any,sg,3,d,0,0' drel='k7t:padZA' posn='160' name='samaya' chunkId='NP6' chunkType='head:NP6'
 +17 padZA VM <fs af='pada,v,any,any,any,,yA,yA' stype='declarative' posn='170' voicetype='active' name='padZA' chunkId='VGF' chunkType='head:VGF'> 
 +18 jaba PRP <fs af='jaba,pn,,,,,,' drel='k7t:kiyA' posn='180' coref='samaya' name='jaba' chunkId='NP7' chunkType='head:NP7'> 
 +19 vahAM PRP <fs af='vahAz,pn,,,,,0_para,' drel='jjmod:wEnAwa' posn='190' vpos='vib_2' name='vahAM' chunkId='NP8' chunkType='head:NP8'> 
 +20 para PSP <fs af='para,psp,,,,,,' posn='200' drel='lwg__psp:vahAM' chunkType='child:NP8' name='para'> 
 +21 wEnAwa JJ <fs af='wEnAwa,adj,any,any,,o,,' drel='nmod:surakRAkarmiyoM' posn='210' name='wEnAwa' chunkId='JJP2' chunkType='head:JJP2'> 
 +22 surakRAkarmiyoM NN <fs af='surakRAkarmI,n,m,pl,3,o,0_ne,0' drel='k1:kiyA' posn='220' vpos='vib_2' name='surakRAkarmiyoM' chunkId='NP9' chunkType='head:NP9'> 
 +23 ne PSP <fs af='ne,psp,,,,,,' posn='230' drel='lwg__psp:surakRAkarmiyoM' chunkType='child:NP9' name='ne'> 
 +24 bOYlIvuda NN <fs af='bOYlIvuda,n,m,sg,3,o,0_kA,0' drel='r6:basu' posn='240' vpos='vib_2' name='bOYlIvuda' chunkId='NP10' chunkType='head:NP10'> 
 +25 kI PSP <fs af='kA,psp,f,sg,,o,,' posn='250' drel='lwg__psp:bOYlIvuda' chunkType='child:NP10' name='kI'
 +26 aBinewrI NN <fs af='aBinewrI,n,f,sg,3,o,0,0' posn='260' drel='nmod:bipASA' chunkType='child:NP11' name='aBinewrI'> 
 +27 bipASA NN <fs af='bipASA,n,f,sg,3,d,0,0' posn='270' drel='nmod:basu' chunkType='child:NP11' name='bipASA'> 
 +28 basu NNP <fs af='basu,n,f,sg,3,o,0_ke_sAWa,0' drel='k2:kiyA' posn='280' vpos='vib_vib_vib_4_5' name='basu' chunkId='NP11' chunkType='head:NP11'> 
 +29 ke PSP <fs af='ke,psp,,,,,,' posn='290' drel='lwg__psp:basu' chunkType='child:NP11' name='ke2'> 
 +30 sAWa NST <fs af='sAWa,nst,m,sg,3,d,,' posn='300' drel='lwg__psp:basu' chunkType='child:NP11' name='sAWa'> 
 +31 xuvyarvahAra NN <fs af='xuvyarvahAra,n,m,sg,3,d,0,0' drel='pof:kiyA' posn='310' name='xuvyarvahAra' chunkId='NP12' chunkType='head:NP12'> 
 +32 kiyA VM <fs af='kara,v,m,sg,any,,yA,yA' drel='nmod__relc:samaya' stype='declarative' posn='320' voicetype='active' name='kiyA' chunkId='VGF2' chunkType='head:VGF2'
 +33 . SYM <fs af='.,punc,,,,,,' posn='330' drel='rsym:kiyA' chunkType='child:VGF2' name='.'>
 </Sentence></code> </Sentence></code>
  

[ Back to the navigation ] [ Back to the content ]