Index of /open-courses-0923/09-CS336-Language-Modeling-from-Scratch/github/assignment5-alignment/tests/_snapshots/
../
test_aggregate_loss_across_microbatch_constant.npz 04-Oct-2026 18:26 268
test_aggregate_loss_across_microbatch_sequence.npz 04-Oct-2026 18:26 268
test_compute_entropy.npz 04-Oct-2026 18:26 344
test_compute_group_normalized_rewards_drgrpo.npz 04-Oct-2026 18:26 582
test_compute_group_normalized_rewards_grpo.npz 04-Oct-2026 18:26 290
test_compute_group_normalized_rewards_maxrl.npz 04-Oct-2026 18:26 280
test_compute_policy_gradient_loss_off_policy.npz 04-Oct-2026 18:26 692
test_compute_policy_gradient_loss_off_policy_gs..> 04-Oct-2026 18:26 344
test_compute_policy_gradient_loss_on_policy.npz 04-Oct-2026 18:26 344
test_compute_rollout_rewards.npz 04-Oct-2026 18:26 296
test_get_response_log_probs.npz 04-Oct-2026 18:26 690
test_grpo_train_step_off_policy[grpo].npz 04-Oct-2026 18:26 8522
test_grpo_train_step_off_policy[gspo].npz 04-Oct-2026 18:26 8522
test_grpo_train_step_off_policy[noclip].npz 04-Oct-2026 18:26 8522
test_grpo_train_step_standard_on_policy.npz 04-Oct-2026 18:26 8522
test_grpo_train_step_variants_on_policy[dr_grpo..> 04-Oct-2026 18:26 8522
test_grpo_train_step_variants_on_policy[grpo_co..> 04-Oct-2026 18:26 8522
test_grpo_train_step_variants_on_policy[maxrl].npz 04-Oct-2026 18:26 8522
test_grpo_train_step_variants_on_policy[rft].npz 04-Oct-2026 18:26 8522
test_masked_normalize_dim0.npz 04-Oct-2026 18:26 4264
test_masked_normalize_dim1.npz 04-Oct-2026 18:26 1064
test_masked_normalize_dimNone.npz 04-Oct-2026 18:26 268
test_masked_normalize_dimlast.npz 04-Oct-2026 18:26 344
test_sft_microbatch_train_step.npz 04-Oct-2026 18:26 620
test_sft_microbatch_train_step_10_steps.npz 04-Oct-2026 18:26 1376
test_sft_microbatch_train_step_normalize.npz 04-Oct-2026 18:26 620
test_tokenize_prompt_and_output.npz 04-Oct-2026 18:26 1233