ireland-news-headlines/solutions/most_common_value_in_train.py
2023-11-16 16:54:06 +01:00

11 lines
355 B
Python

import pandas as pd
r_out = pd.read_csv('../train/expected.tsv', names = ('class',))
most_common = r_out['class'].value_counts().idxmax()
for dataset in 'dev-0', 'test-A', 'test-B':
with open(f'../{dataset}/out.tsv', 'w') as f_out, open(f'../{dataset}/in.tsv', 'r') as f_in:
for line_in in f_in:
f_out.write(most_common + '\n')