Repository navigation
Expand file tree
/
Copy path_toctree.yml
More file actions
133 lines (133 loc) · 3.13 KB
/
Copy path_toctree.yml
File metadata and controls
133 lines (133 loc) · 3.13 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
- sections:
- local: index
title: TRL
- local: installation
title: Installation
- local: quickstart
title: Quickstart
- local: usage_stats
title: Usage Stats Collection
title: Getting started
- sections:
- local: chat_templates
title: Chat Templates
- local: dataset_formats
title: Dataset Formats
- local: paper_index
title: Paper Index
title: Conceptual Guides
- sections: # Sorted alphabetically
- local: distillation_trainer
title: Distillation
- local: dpo_trainer
title: DPO
- local: grpo_trainer
title: GRPO
- local: kto_trainer
title: KTO
- local: reward_trainer
title: Reward
- local: rloo_trainer
title: RLOO
- local: sft_trainer
title: SFT
title: Trainers
- sections:
- local: clis
title: Command Line Interface (CLI)
- local: jobs_training
title: Training using Jobs
- local: customization
title: Customizing the Training
- local: reducing_memory_usage
title: Reducing Memory Usage
- local: speeding_up_training
title: Speeding Up Training
- local: distributing_training
title: Distributing Training
- local: long_context_training
title: Training Beyond 1M Tokens
- local: use_model
title: Using Trained Models
title: How-to guides
- sections:
- local: deepspeed_integration
title: DeepSpeed
- local: harbor
title: Harbor
- local: kernels_hub
title: Kernels Hub
- local: openenv
title: OpenEnv
- local: openreward
title: OpenReward
- local: peft_integration
title: PEFT
- local: rapidfire_integration
title: RapidFire AI
- local: trackio_integration
title: Trackio
- local: unsloth_integration
title: Unsloth
- local: vllm_integration
title: vLLM
title: Integrations
- sections:
- local: example_overview
title: Example Overview
- local: community_tutorials
title: Community Tutorials
- local: lora_without_regret
title: LoRA Without Regret
title: Examples
- sections:
- sections:
- local: chat_template_utils
title: Chat Template Utilities
- local: data_utils
title: Data Utilities
- local: script_utils
title: Script Utilities
title: Utilities
- local: callbacks
title: Callbacks
- local: rewards
title: Reward Functions
title: API
- sections:
- local: experimental_overview
title: Experimental Overview
- local: a2po_trainer # Sorted alphabetically
title: A2PO
- local: async_distillation_trainer
title: Async Distillation
- local: async_grpo_trainer
title: Asynchronous GRPO
- local: bema_for_reference_model
title: BEMA for Reference Model
- local: cpo_trainer
title: CPO
- local: gkd_trainer
title: GKD
- local: gmpo
title: GMPO
- local: gold_trainer
title: GOLD
- local: iw_opd_trainer
title: IW-OPD
- local: merge_model_callback
title: MergeModelCallback
- local: online_dpo_trainer
title: Online DPO
- local: orpo_trainer
title: ORPO
- local: sdft_trainer
title: SDFT
- local: sdpo_trainer
title: SDPO
- local: ssd_trainer
title: SSD
- local: tpo_trainer
title: TPO
title: Experimental
isExpanded: false