"""
33."A's B" Permalink
Extract a noun phrase in which two nouns are connected by "no"
[[{'surface': '', 'base': '*', 'pos': 'BOS/EOS', 'pos1': '*'},
{'surface': 'one', 'base': 'one', 'pos': 'noun', 'pos1': 'number'},
{'surface': '', 'base': '*', 'pos': 'BOS/EOS', 'pos1': '*'}],
[{'surface': '', 'base': '*', 'pos': 'BOS/EOS', 'pos1': '*'},
{'surface': 'I', 'base': 'I', 'pos': 'noun', 'pos1': '代noun'},
{'surface': 'Is', 'base': 'Is', 'pos': 'Particle', 'pos1': '係Particle'},
{'surface': 'Cat', 'base': 'Cat', 'pos': 'noun', 'pos1': 'General'},
{'surface': 'so', 'base': 'Is', 'pos': 'Auxiliary verb', 'pos1': '*'},
{'surface': 'is there', 'base': 'is there', 'pos': 'Auxiliary verb', 'pos1': '*'},
{'surface': '。', 'base': '。', 'pos': 'symbol', 'pos1': 'Kuten'},
{'surface': '', 'base': '*', 'pos': 'BOS/EOS', 'pos1': '*'}],
"""
import itertools
from typing import List
import utils
def get_none_phase(sentence: List[dict]) -> List[str]:
result = []
for i, word in enumerate(sentence):
if (
word["surface"] == "of"
and sentence[i - 1]["pos"] == "noun"
and sentence[i + 1]["pos"] == "noun"
):
result.append(sentence[i - 1]["surface"] + "of" + sentence[i + 1]["surface"])
return result
data = utils.read_json("30_neko_mecab.json")
none_phases = [get_none_phase(sentence) for sentence in data]
# In [75]: none_phases[:10]
# Out[75]: [[], [], [], [], [], [], [], [], [], ['His palm']]
flat = list(itertools.chain(*none_phases))
# ['His palm', 'On the palm', 'Student's face', 'Should face', 'In the middle of the face']
Recommended Posts