100knock #47
123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249
package main
import (
"./fileio"
"fmt"
"os"
"sort"
"strconv"
"strings"
)
type Morph struct {
surface string
base string
pos string
pos1 string
}
type Chunk struct {
morphs []Morph
dst int
srcs []int
}
type wordSet struct {
phrase string
particle string
}
func parseArticle(lines []string) [][]Chunk {
article := make([][]Chunk, 0)
morphemes := make([]Morph, 0)
sentence := make([]Chunk, 0)
var chunk Chunk
for _, line := range lines {
if line == "EOS" {
if len(morphemes) > 0 {
chunk.morphs = morphemes
sentence = append(sentence, chunk)
morphemes = make([]Morph, 0)
}
if len(sentence) > 0 {
initSourceIndex(sentence)
article = append(article, sentence)
sentence = make([]Chunk, 0)
}
continue
}
if line[0] == '*' {
if len(morphemes) > 0 {
chunk.morphs = morphemes
sentence = append(sentence, chunk)
morphemes = make([]Morph, 0)
}
//* 文節番号 係り先の文節番号(係り先なし:-1) 主辞の形態素番号/機能語の形態素番号 係り関係のスコア
words := strings.Split(line, " ")
// Remove "D"
dst, err := strconv.Atoi(words[2][:len(words[2])-1])
if err != nil {
panic(err)
}
chunk = initChunk(chunk, dst)
continue
}
//表層形\t品詞,品詞細分類1,品詞細分類2,品詞細分類3,活用形,活用型,原形,読み,発音
word := strings.Split(line, "\t")
words := strings.Split(word[1], ",")
morpheme := Morph{
surface: word[0],
base: words[6],
pos: words[0],
pos1: words[1],
}
morphemes = append(morphemes, morpheme)
}
return article
}
func initChunk(chunk Chunk, dst int) Chunk {
chunk.morphs = make([]Morph, 0)
chunk.dst = dst
chunk.srcs = make([]int, 0)
return chunk
}
func initSourceIndex(sentence []Chunk) {
for i, chunk := range sentence {
if chunk.dst < 0 {
continue
}
sentence[chunk.dst].srcs = append(sentence[chunk.dst].srcs, i)
}
}
func containVerb(morphs []Morph) bool {
for _, morph := range morphs {
if morph.pos == "動詞" {
return true
}
}
return false
}
func containParticle(morphs []Morph) bool {
for _, morph := range morphs {
if morph.pos == "助詞" {
return true
}
}
return false
}
func containParticleWo(morphs []Morph) bool {
for _, morph := range morphs {
if morph.pos == "助詞" && morph.surface == "を" {
return true
}
}
return false
}
func fetchVerb(morphs []Morph) string {
for _, morph := range morphs {
if morph.pos == "動詞" {
return morph.base
}
}
return ""
}
func fetchParticleAll(morphs []Morph) string {
particles := ""
for _, morph := range morphs {
if morph.pos == "助詞" {
particles += morph.base + " "
}
}
return strings.TrimRight(particles, " ")
}
func fetchSahenSetsuzokuNounWo(morphs []Morph) string {
for i, morph := range morphs {
if morph.pos != "名詞" || morph.pos1 != "サ変接続" {
continue
}
if i+1 < len(morphs) {
if morphs[i+1].surface == "を" {
return morph.surface + "を"
}
}
}
return ""
}
func joinPhrase(morphs []Morph) string {
joinString := ""
for _, morph := range morphs {
if morph.pos == "記号" {
continue
}
joinString += morph.surface
}
return joinString
}
func printVerbCases(sentence []Chunk) {
for _, chunk := range sentence {
if containVerb(chunk.morphs) == false {
continue
}
sahenNounWo := ""
for _, index := range chunk.srcs {
sahenNounWo = fetchSahenSetsuzokuNounWo(sentence[index].morphs)
if sahenNounWo == "" {
continue
}
predicate := sahenNounWo + fetchVerb(chunk.morphs)
wordSets := make([]wordSet, 0)
var set wordSet
for _, i := range chunk.srcs {
if i == index {
continue
}
if containParticle(sentence[i].morphs) {
set.particle = fetchParticleAll(sentence[i].morphs)
set.phrase = joinPhrase(sentence[i].morphs)
wordSets = append(wordSets, set)
}
}
sort.SliceStable(wordSets, func(i, j int) bool {
return wordSets[i].particle < wordSets[j].particle
})
particles := ""
phrases := ""
for _, set := range wordSets {
particles += set.particle + " "
phrases += set.phrase + " "
}
if particles != "" {
particles = strings.TrimRight(particles, " ")
phrases = strings.TrimRight(phrases, " ")
fmt.Printf("%s\t%s\t%s\n", predicate, particles, phrases)
}
}
}
}
func main() {
if len(os.Args) != 2 {
fmt.Println("Usage: main <filepath>")
os.Exit(1)
}
lines := fileio.ReadFileAllLines(os.Args[1])
article := parseArticle(lines)
for _, sentence := range article {
printVerbCases(sentence)
}
}
Go
INFO