tokenized_dataset_train = LineByLineTextDataset( tokenizer= tokenizer, raw_datasets = dataset, max_length=256, )