| 11 | # The template prompt dataset class that all new dataset porting needs to |
| 12 | # follow in order to have a unified API and unified data format. |
| 13 | class PromptRawDataset(object): |
| 14 | |
| 15 | def __init__(self, output_path, seed, local_rank, dataset_name): |
| 16 | self.output_path = output_path |
| 17 | self.seed = seed |
| 18 | self.local_rank = local_rank |
| 19 | # default load from disk |
| 20 | if "Anthropic/hh-rlhf" in dataset_name: |
| 21 | self.raw_datasets = load_from_disk(dataset_name) |
| 22 | |
| 23 | def get_train_data(self): |
| 24 | return |
| 25 | |
| 26 | def get_eval_data(self): |
| 27 | return |
| 28 | |
| 29 | # The prompt should be in the format of: " Human: " + actual_prompt_sentence + " Assistant:" |
| 30 | def get_prompt(self, sample): |
| 31 | return |
| 32 | |
| 33 | # The chosen response should be in the format of: " " + actual_response_sentence |
| 34 | def get_answer(self, sample): |
| 35 | return |
| 36 | |
| 37 | def get_prompt_and_answer(self, sample): |
| 38 | return |
| 39 | |
| 40 | |
| 41 |
nothing calls this directly
no outgoing calls
no test coverage detected