Skip to content

launch

topic_segmentation.launch

command(model, dataset, data_dir, profile, objective, output, cache, max_length=None, max_steps=None)

Source code in src/topic_segmentation/launch.py
 8
 9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
def command(model, dataset, data_dir, profile, objective, output, cache, max_length=None, max_steps=None):
    arguments = [
        sys.executable,
        "-m",
        "torch.distributed.run",
        "--nproc_per_node=1",
        "--module",
        "topic_segmentation.training",
        "--model_name_or_path", str(model),
        "--dataset_name", dataset,
        "--data_dir", str(data_dir),
        "--dataset_cache_dir", str(cache),
        "--do_train", "True",
        "--do_eval", "True",
        "--do_predict", "True",
        "--seed", str(profile["seed"]),
        "--max_seq_length", str(max_length or profile["max_sequence_length"]),
        "--learning_rate", str(profile["learning_rate"]),
        "--per_device_train_batch_size", str(profile["per_device_batch_size"]),
        "--gradient_accumulation_steps", str(profile["gradient_accumulation_steps"]),
        "--per_device_eval_batch_size", str(profile["per_device_batch_size"]),
        "--eval_strategy", "steps",
        "--eval_cnt", str(profile["evaluation_points"]),
        "--load_best_model_at_end", "True",
        "--save_total_limit", "2",
        "--metric_for_best_model", "overall_f1",
        "--eval_accumulation_steps", "1000",
        "--preprocessing_num_workers", "5",
        "--gradient_checkpointing", "False",
        "--do_da_ts", str(objective["do_da_ts"]),
        "--do_tssp", str(objective["do_tssp"]),
        "--ts_loss_weight", "1.0",
        "--cl_loss_weight", str(objective["cl_loss_weight"]),
        "--cl_temp", str(objective.get("cl_temp", 0.1)),
        "--cl_positive_k", "1",
        "--cl_negative_k", "3",
        "--tssp_loss_weight", str(objective["tssp_loss_weight"]),
        "--output_dir", str(output),
    ]
    if max_steps:
        return arguments + ["--max_steps", str(max_steps)]
    return arguments + ["--num_train_epochs", str(profile["epochs"])]

environment(gpu)

Configure one visible GPU and local caches; disable torch.compile.

Source code in src/topic_segmentation/launch.py
52
53
54
55
56
57
58
59
60
61
62
63
64
def environment(gpu):
    """Configure one visible GPU and local caches; disable torch.compile."""
    variables = dict(
        os.environ,
        CUDA_VISIBLE_DEVICES=str(gpu),
        HF_MODULES_CACHE=str(REPOSITORY / "cache/hf_modules"),
        HF_DATASETS_CACHE=str(REPOSITORY / "cache/hf_datasets"),
        TMPDIR=str(REPOSITORY / "cache/tmp"),
        TORCHDYNAMO_DISABLE="1",
        OMP_NUM_THREADS="8",
    )
    Path(variables["TMPDIR"]).mkdir(parents=True, exist_ok=True)
    return variables