# OMEGA Hard L6 mini-pilot — 30 problems × 8 reps. Level 6 (above 7B "exhard").
task: omega_hard_l6_n30_pass8
dataset_path: parquet
dataset_kwargs:
  data_files:
    test: data/omega/hard_explorative_l6_n30.parquet
output_type: generate_until
test_split: test
doc_to_text: "Problem: {{question}}\n\nPlease reason step by step, and put your final answer within \\boxed{}."
doc_to_target: "{{answer}}"
process_results: !function utils.process_results
generation_kwargs:
  until: ["Problem:", "</s>", "<|im_end|>", "<|endoftext|>", "<|end▁of▁sentence|>"]
  do_sample: true
  temperature: 0.6
  top_p: 0.95
  max_gen_toks: 25000
repeats: 8
metric_list:
  - metric: exact_match
    aggregation: mean
    higher_is_better: true
num_fewshot: 0
metadata:
  version: 1.0
  description: "OMEGA Hard L6 mini-pilot N=30 × 8."
