diff --git a/.DS_Store b/.DS_Store new file mode 100644 index 00000000..45b2c008 Binary files /dev/null and b/.DS_Store differ diff --git a/_sources/api/ml.rl.evaluation.rst.txt b/_sources/api/ml.rl.evaluation.rst.txt new file mode 100644 index 00000000..b8b2c409 --- /dev/null +++ b/_sources/api/ml.rl.evaluation.rst.txt @@ -0,0 +1,85 @@ +ml.rl.evaluation package +======================== + +Submodules +---------- + +ml.rl.evaluation.cpe module +--------------------------- + +.. automodule:: ml.rl.evaluation.cpe + :members: + :undoc-members: + :show-inheritance: + +ml.rl.evaluation.doubly\_robust\_estimator module +------------------------------------------------- + +.. automodule:: ml.rl.evaluation.doubly_robust_estimator + :members: + :undoc-members: + :show-inheritance: + +ml.rl.evaluation.evaluation\_data\_page module +---------------------------------------------- + +.. automodule:: ml.rl.evaluation.evaluation_data_page + :members: + :undoc-members: + :show-inheritance: + +ml.rl.evaluation.evaluator module +--------------------------------- + +.. automodule:: ml.rl.evaluation.evaluator + :members: + :undoc-members: + :show-inheritance: + +ml.rl.evaluation.ranking\_evaluator module +------------------------------------------ + +.. automodule:: ml.rl.evaluation.ranking_evaluator + :members: + :undoc-members: + :show-inheritance: + +ml.rl.evaluation.reward\_net\_evaluator module +---------------------------------------------- + +.. automodule:: ml.rl.evaluation.reward_net_evaluator + :members: + :undoc-members: + :show-inheritance: + +ml.rl.evaluation.sequential\_doubly\_robust\_estimator module +------------------------------------------------------------- + +.. automodule:: ml.rl.evaluation.sequential_doubly_robust_estimator + :members: + :undoc-members: + :show-inheritance: + +ml.rl.evaluation.weighted\_sequential\_doubly\_robust\_estimator module +----------------------------------------------------------------------- + +.. automodule:: ml.rl.evaluation.weighted_sequential_doubly_robust_estimator + :members: + :undoc-members: + :show-inheritance: + +ml.rl.evaluation.world\_model\_evaluator module +----------------------------------------------- + +.. automodule:: ml.rl.evaluation.world_model_evaluator + :members: + :undoc-members: + :show-inheritance: + +Module contents +--------------- + +.. automodule:: ml.rl.evaluation + :members: + :undoc-members: + :show-inheritance: diff --git a/_sources/api/ml.rl.models.rst.txt b/_sources/api/ml.rl.models.rst.txt new file mode 100644 index 00000000..f4eda389 --- /dev/null +++ b/_sources/api/ml.rl.models.rst.txt @@ -0,0 +1,157 @@ +ml.rl.models package +==================== + +Submodules +---------- + +ml.rl.models.actor module +------------------------- + +.. automodule:: ml.rl.models.actor + :members: + :undoc-members: + :show-inheritance: + +ml.rl.models.base module +------------------------ + +.. automodule:: ml.rl.models.base + :members: + :undoc-members: + :show-inheritance: + +ml.rl.models.bcq module +----------------------- + +.. automodule:: ml.rl.models.bcq + :members: + :undoc-members: + :show-inheritance: + +ml.rl.models.categorical\_dqn module +------------------------------------ + +.. automodule:: ml.rl.models.categorical_dqn + :members: + :undoc-members: + :show-inheritance: + +ml.rl.models.cem\_planner module +-------------------------------- + +.. automodule:: ml.rl.models.cem_planner + :members: + :undoc-members: + :show-inheritance: + +ml.rl.models.convolutional\_network module +------------------------------------------ + +.. automodule:: ml.rl.models.convolutional_network + :members: + :undoc-members: + :show-inheritance: + +ml.rl.models.dqn module +----------------------- + +.. automodule:: ml.rl.models.dqn + :members: + :undoc-members: + :show-inheritance: + +ml.rl.models.dueling\_q\_network module +--------------------------------------- + +.. automodule:: ml.rl.models.dueling_q_network + :members: + :undoc-members: + :show-inheritance: + +ml.rl.models.dueling\_quantile\_dqn module +------------------------------------------ + +.. automodule:: ml.rl.models.dueling_quantile_dqn + :members: + :undoc-members: + :show-inheritance: + +ml.rl.models.example\_sequence\_model module +-------------------------------------------- + +.. automodule:: ml.rl.models.example_sequence_model + :members: + :undoc-members: + :show-inheritance: + +ml.rl.models.fully\_connected\_network module +--------------------------------------------- + +.. automodule:: ml.rl.models.fully_connected_network + :members: + :undoc-members: + :show-inheritance: + +ml.rl.models.mdn\_rnn module +---------------------------- + +.. automodule:: ml.rl.models.mdn_rnn + :members: + :undoc-members: + :show-inheritance: + +ml.rl.models.no\_soft\_update\_embedding module +----------------------------------------------- + +.. automodule:: ml.rl.models.no_soft_update_embedding + :members: + :undoc-members: + :show-inheritance: + +ml.rl.models.parametric\_dqn module +----------------------------------- + +.. automodule:: ml.rl.models.parametric_dqn + :members: + :undoc-members: + :show-inheritance: + +ml.rl.models.quantile\_dqn module +--------------------------------- + +.. automodule:: ml.rl.models.quantile_dqn + :members: + :undoc-members: + :show-inheritance: + +ml.rl.models.seq2slate module +----------------------------- + +.. automodule:: ml.rl.models.seq2slate + :members: + :undoc-members: + :show-inheritance: + +ml.rl.models.seq2slate\_reward module +------------------------------------- + +.. automodule:: ml.rl.models.seq2slate_reward + :members: + :undoc-members: + :show-inheritance: + +ml.rl.models.world\_model module +-------------------------------- + +.. automodule:: ml.rl.models.world_model + :members: + :undoc-members: + :show-inheritance: + +Module contents +--------------- + +.. automodule:: ml.rl.models + :members: + :undoc-members: + :show-inheritance: diff --git a/_sources/api/ml.rl.polyfill.rst.txt b/_sources/api/ml.rl.polyfill.rst.txt new file mode 100644 index 00000000..0ef966a2 --- /dev/null +++ b/_sources/api/ml.rl.polyfill.rst.txt @@ -0,0 +1,37 @@ +ml.rl.polyfill package +====================== + +Submodules +---------- + +ml.rl.polyfill.decorators module +-------------------------------- + +.. automodule:: ml.rl.polyfill.decorators + :members: + :undoc-members: + :show-inheritance: + +ml.rl.polyfill.exceptions module +-------------------------------- + +.. automodule:: ml.rl.polyfill.exceptions + :members: + :undoc-members: + :show-inheritance: + +ml.rl.polyfill.types module +--------------------------- + +.. automodule:: ml.rl.polyfill.types + :members: + :undoc-members: + :show-inheritance: + +Module contents +--------------- + +.. automodule:: ml.rl.polyfill + :members: + :undoc-members: + :show-inheritance: diff --git a/_sources/api/ml.rl.prediction.rst.txt b/_sources/api/ml.rl.prediction.rst.txt new file mode 100644 index 00000000..551c92c9 --- /dev/null +++ b/_sources/api/ml.rl.prediction.rst.txt @@ -0,0 +1,29 @@ +ml.rl.prediction package +======================== + +Submodules +---------- + +ml.rl.prediction.dqn\_torch\_predictor module +--------------------------------------------- + +.. automodule:: ml.rl.prediction.dqn_torch_predictor + :members: + :undoc-members: + :show-inheritance: + +ml.rl.prediction.predictor\_wrapper module +------------------------------------------ + +.. automodule:: ml.rl.prediction.predictor_wrapper + :members: + :undoc-members: + :show-inheritance: + +Module contents +--------------- + +.. automodule:: ml.rl.prediction + :members: + :undoc-members: + :show-inheritance: diff --git a/_sources/api/ml.rl.preprocessing.rst.txt b/_sources/api/ml.rl.preprocessing.rst.txt new file mode 100644 index 00000000..61fe4a1b --- /dev/null +++ b/_sources/api/ml.rl.preprocessing.rst.txt @@ -0,0 +1,61 @@ +ml.rl.preprocessing package +=========================== + +Submodules +---------- + +ml.rl.preprocessing.batch\_preprocessor module +---------------------------------------------- + +.. automodule:: ml.rl.preprocessing.batch_preprocessor + :members: + :undoc-members: + :show-inheritance: + +ml.rl.preprocessing.identify\_types module +------------------------------------------ + +.. automodule:: ml.rl.preprocessing.identify_types + :members: + :undoc-members: + :show-inheritance: + +ml.rl.preprocessing.normalization module +---------------------------------------- + +.. automodule:: ml.rl.preprocessing.normalization + :members: + :undoc-members: + :show-inheritance: + +ml.rl.preprocessing.postprocessor module +---------------------------------------- + +.. automodule:: ml.rl.preprocessing.postprocessor + :members: + :undoc-members: + :show-inheritance: + +ml.rl.preprocessing.preprocessor module +--------------------------------------- + +.. automodule:: ml.rl.preprocessing.preprocessor + :members: + :undoc-members: + :show-inheritance: + +ml.rl.preprocessing.sparse\_to\_dense module +-------------------------------------------- + +.. automodule:: ml.rl.preprocessing.sparse_to_dense + :members: + :undoc-members: + :show-inheritance: + +Module contents +--------------- + +.. automodule:: ml.rl.preprocessing + :members: + :undoc-members: + :show-inheritance: diff --git a/_sources/api/ml.rl.readers.rst.txt b/_sources/api/ml.rl.readers.rst.txt new file mode 100644 index 00000000..a81f6f7a --- /dev/null +++ b/_sources/api/ml.rl.readers.rst.txt @@ -0,0 +1,45 @@ +ml.rl.readers package +===================== + +Submodules +---------- + +ml.rl.readers.base module +------------------------- + +.. automodule:: ml.rl.readers.base + :members: + :undoc-members: + :show-inheritance: + +ml.rl.readers.data\_streamer module +----------------------------------- + +.. automodule:: ml.rl.readers.data_streamer + :members: + :undoc-members: + :show-inheritance: + +ml.rl.readers.json\_dataset\_reader module +------------------------------------------ + +.. automodule:: ml.rl.readers.json_dataset_reader + :members: + :undoc-members: + :show-inheritance: + +ml.rl.readers.nparray\_reader module +------------------------------------ + +.. automodule:: ml.rl.readers.nparray_reader + :members: + :undoc-members: + :show-inheritance: + +Module contents +--------------- + +.. automodule:: ml.rl.readers + :members: + :undoc-members: + :show-inheritance: diff --git a/_sources/api/ml.rl.rst.txt b/_sources/api/ml.rl.rst.txt new file mode 100644 index 00000000..6d8dca25 --- /dev/null +++ b/_sources/api/ml.rl.rst.txt @@ -0,0 +1,85 @@ +ml.rl package +============= + +Subpackages +----------- + +.. toctree:: + :maxdepth: 4 + + ml.rl.evaluation + ml.rl.models + ml.rl.polyfill + ml.rl.prediction + ml.rl.preprocessing + ml.rl.readers + ml.rl.simulators + ml.rl.training + ml.rl.workflow + +Submodules +---------- + +ml.rl.caffe\_utils module +------------------------- + +.. automodule:: ml.rl.caffe_utils + :members: + :undoc-members: + :show-inheritance: + +ml.rl.debug\_on\_error module +----------------------------- + +.. automodule:: ml.rl.debug_on_error + :members: + :undoc-members: + :show-inheritance: + +ml.rl.json\_serialize module +---------------------------- + +.. automodule:: ml.rl.json_serialize + :members: + :undoc-members: + :show-inheritance: + +ml.rl.parameters module +----------------------- + +.. automodule:: ml.rl.parameters + :members: + :undoc-members: + :show-inheritance: + +ml.rl.tensorboardX module +------------------------- + +.. automodule:: ml.rl.tensorboardX + :members: + :undoc-members: + :show-inheritance: + +ml.rl.torch\_utils module +------------------------- + +.. automodule:: ml.rl.torch_utils + :members: + :undoc-members: + :show-inheritance: + +ml.rl.types module +------------------ + +.. automodule:: ml.rl.types + :members: + :undoc-members: + :show-inheritance: + +Module contents +--------------- + +.. automodule:: ml.rl + :members: + :undoc-members: + :show-inheritance: diff --git a/_sources/api/ml.rl.simulators.rst.txt b/_sources/api/ml.rl.simulators.rst.txt new file mode 100644 index 00000000..fd0b0364 --- /dev/null +++ b/_sources/api/ml.rl.simulators.rst.txt @@ -0,0 +1,21 @@ +ml.rl.simulators package +======================== + +Submodules +---------- + +ml.rl.simulators.recsim module +------------------------------ + +.. automodule:: ml.rl.simulators.recsim + :members: + :undoc-members: + :show-inheritance: + +Module contents +--------------- + +.. automodule:: ml.rl.simulators + :members: + :undoc-members: + :show-inheritance: diff --git a/_sources/api/ml.rl.training.gradient_free.rst.txt b/_sources/api/ml.rl.training.gradient_free.rst.txt new file mode 100644 index 00000000..63d02ad8 --- /dev/null +++ b/_sources/api/ml.rl.training.gradient_free.rst.txt @@ -0,0 +1,29 @@ +ml.rl.training.gradient\_free package +===================================== + +Submodules +---------- + +ml.rl.training.gradient\_free.es\_worker module +----------------------------------------------- + +.. automodule:: ml.rl.training.gradient_free.es_worker + :members: + :undoc-members: + :show-inheritance: + +ml.rl.training.gradient\_free.evolution\_pool module +---------------------------------------------------- + +.. automodule:: ml.rl.training.gradient_free.evolution_pool + :members: + :undoc-members: + :show-inheritance: + +Module contents +--------------- + +.. automodule:: ml.rl.training.gradient_free + :members: + :undoc-members: + :show-inheritance: diff --git a/_sources/api/ml.rl.training.ranking.rst.txt b/_sources/api/ml.rl.training.ranking.rst.txt new file mode 100644 index 00000000..7cd69c56 --- /dev/null +++ b/_sources/api/ml.rl.training.ranking.rst.txt @@ -0,0 +1,29 @@ +ml.rl.training.ranking package +============================== + +Submodules +---------- + +ml.rl.training.ranking.seq2slate\_tf\_trainer module +---------------------------------------------------- + +.. automodule:: ml.rl.training.ranking.seq2slate_tf_trainer + :members: + :undoc-members: + :show-inheritance: + +ml.rl.training.ranking.seq2slate\_trainer module +------------------------------------------------ + +.. automodule:: ml.rl.training.ranking.seq2slate_trainer + :members: + :undoc-members: + :show-inheritance: + +Module contents +--------------- + +.. automodule:: ml.rl.training.ranking + :members: + :undoc-members: + :show-inheritance: diff --git a/_sources/api/ml.rl.training.rst.txt b/_sources/api/ml.rl.training.rst.txt new file mode 100644 index 00000000..8b485924 --- /dev/null +++ b/_sources/api/ml.rl.training.rst.txt @@ -0,0 +1,159 @@ +ml.rl.training package +====================== + +Subpackages +----------- + +.. toctree:: + :maxdepth: 4 + + ml.rl.training.gradient_free + ml.rl.training.ranking + ml.rl.training.world_model + +Submodules +---------- + +ml.rl.training.c51\_trainer module +---------------------------------- + +.. automodule:: ml.rl.training.c51_trainer + :members: + :undoc-members: + :show-inheritance: + +ml.rl.training.cem\_trainer module +---------------------------------- + +.. automodule:: ml.rl.training.cem_trainer + :members: + :undoc-members: + :show-inheritance: + +ml.rl.training.dqn\_trainer module +---------------------------------- + +.. automodule:: ml.rl.training.dqn_trainer + :members: + :undoc-members: + :show-inheritance: + +ml.rl.training.dqn\_trainer\_base module +---------------------------------------- + +.. automodule:: ml.rl.training.dqn_trainer_base + :members: + :undoc-members: + :show-inheritance: + +ml.rl.training.imitator\_training module +---------------------------------------- + +.. automodule:: ml.rl.training.imitator_training + :members: + :undoc-members: + :show-inheritance: + +ml.rl.training.loss\_reporter module +------------------------------------ + +.. automodule:: ml.rl.training.loss_reporter + :members: + :undoc-members: + :show-inheritance: + +ml.rl.training.on\_policy\_predictor module +------------------------------------------- + +.. automodule:: ml.rl.training.on_policy_predictor + :members: + :undoc-members: + :show-inheritance: + +ml.rl.training.parametric\_dqn\_trainer module +---------------------------------------------- + +.. automodule:: ml.rl.training.parametric_dqn_trainer + :members: + :undoc-members: + :show-inheritance: + +ml.rl.training.qrdqn\_trainer module +------------------------------------ + +.. automodule:: ml.rl.training.qrdqn_trainer + :members: + :undoc-members: + :show-inheritance: + +ml.rl.training.reward\_network\_trainer module +---------------------------------------------- + +.. automodule:: ml.rl.training.reward_network_trainer + :members: + :undoc-members: + :show-inheritance: + +ml.rl.training.rl\_dataset module +--------------------------------- + +.. automodule:: ml.rl.training.rl_dataset + :members: + :undoc-members: + :show-inheritance: + +ml.rl.training.rl\_trainer\_pytorch module +------------------------------------------ + +.. automodule:: ml.rl.training.rl_trainer_pytorch + :members: + :undoc-members: + :show-inheritance: + +ml.rl.training.sac\_trainer module +---------------------------------- + +.. automodule:: ml.rl.training.sac_trainer + :members: + :undoc-members: + :show-inheritance: + +ml.rl.training.slate\_q\_trainer module +--------------------------------------- + +.. automodule:: ml.rl.training.slate_q_trainer + :members: + :undoc-members: + :show-inheritance: + +ml.rl.training.td3\_trainer module +---------------------------------- + +.. automodule:: ml.rl.training.td3_trainer + :members: + :undoc-members: + :show-inheritance: + +ml.rl.training.trainer module +----------------------------- + +.. automodule:: ml.rl.training.trainer + :members: + :undoc-members: + :show-inheritance: + +ml.rl.training.training\_data\_page module +------------------------------------------ + +.. automodule:: ml.rl.training.training_data_page + :members: + :undoc-members: + :show-inheritance: + +Module contents +--------------- + +.. automodule:: ml.rl.training + :members: + :undoc-members: + :show-inheritance: diff --git a/_sources/api/ml.rl.training.world_model.rst.txt b/_sources/api/ml.rl.training.world_model.rst.txt new file mode 100644 index 00000000..3797929c --- /dev/null +++ b/_sources/api/ml.rl.training.world_model.rst.txt @@ -0,0 +1,21 @@ +ml.rl.training.world\_model package +=================================== + +Submodules +---------- + +ml.rl.training.world\_model.mdnrnn\_trainer module +-------------------------------------------------- + +.. automodule:: ml.rl.training.world_model.mdnrnn_trainer + :members: + :undoc-members: + :show-inheritance: + +Module contents +--------------- + +.. automodule:: ml.rl.training.world_model + :members: + :undoc-members: + :show-inheritance: diff --git a/_sources/api/ml.rl.workflow.rst.txt b/_sources/api/ml.rl.workflow.rst.txt new file mode 100644 index 00000000..2db46872 --- /dev/null +++ b/_sources/api/ml.rl.workflow.rst.txt @@ -0,0 +1,77 @@ +ml.rl.workflow package +====================== + +Submodules +---------- + +ml.rl.workflow.base\_workflow module +------------------------------------ + +.. automodule:: ml.rl.workflow.base_workflow + :members: + :undoc-members: + :show-inheritance: + +ml.rl.workflow.create\_normalization\_metadata module +----------------------------------------------------- + +.. automodule:: ml.rl.workflow.create_normalization_metadata + :members: + :undoc-members: + :show-inheritance: + +ml.rl.workflow.dqn\_workflow module +----------------------------------- + +.. automodule:: ml.rl.workflow.dqn_workflow + :members: + :undoc-members: + :show-inheritance: + +ml.rl.workflow.helpers module +----------------------------- + +.. automodule:: ml.rl.workflow.helpers + :members: + :undoc-members: + :show-inheritance: + +ml.rl.workflow.page\_handler module +----------------------------------- + +.. automodule:: ml.rl.workflow.page_handler + :members: + :undoc-members: + :show-inheritance: + +ml.rl.workflow.parametric\_dqn\_workflow module +----------------------------------------------- + +.. automodule:: ml.rl.workflow.parametric_dqn_workflow + :members: + :undoc-members: + :show-inheritance: + +ml.rl.workflow.preprocess\_handler module +----------------------------------------- + +.. automodule:: ml.rl.workflow.preprocess_handler + :members: + :undoc-members: + :show-inheritance: + +ml.rl.workflow.transitional module +---------------------------------- + +.. automodule:: ml.rl.workflow.transitional + :members: + :undoc-members: + :show-inheritance: + +Module contents +--------------- + +.. automodule:: ml.rl.workflow + :members: + :undoc-members: + :show-inheritance: diff --git a/_sources/api/ml.rst.txt b/_sources/api/ml.rst.txt new file mode 100644 index 00000000..fd94011c --- /dev/null +++ b/_sources/api/ml.rst.txt @@ -0,0 +1,18 @@ +ml package +========== + +Subpackages +----------- + +.. toctree:: + :maxdepth: 4 + + ml.rl + +Module contents +--------------- + +.. automodule:: ml + :members: + :undoc-members: + :show-inheritance: diff --git a/_sources/api/reagent.training.cb.rst.txt b/_sources/api/reagent.training.cb.rst.txt new file mode 100644 index 00000000..6484d74c --- /dev/null +++ b/_sources/api/reagent.training.cb.rst.txt @@ -0,0 +1,21 @@ +reagent.training.cb package +=========================== + +Submodules +---------- + +reagent.training.cb.linucb\_trainer module +------------------------------------------ + +.. automodule:: reagent.training.cb.linucb_trainer + :members: + :undoc-members: + :show-inheritance: + +Module contents +--------------- + +.. automodule:: reagent.training.cb + :members: + :undoc-members: + :show-inheritance: diff --git a/_static/.DS_Store b/_static/.DS_Store new file mode 100644 index 00000000..f113a79e Binary files /dev/null and b/_static/.DS_Store differ diff --git a/api/ml.html b/api/ml.html new file mode 100644 index 00000000..0abdf1a6 --- /dev/null +++ b/api/ml.html @@ -0,0 +1,278 @@ + + + + + + + ml package — ReAgent 1.0 documentation + + + + + + + + + + + + + + + + +
+ + +
+ +
+
+
+ +
+
+
+
+ +
+

ml package

+
+

Subpackages

+
+ +
+
+
+

Module contents

+
+
+ + +
+
+ +
+
+
+
+ + + + \ No newline at end of file diff --git a/api/ml.rl.evaluation.html b/api/ml.rl.evaluation.html new file mode 100644 index 00000000..2c8fb075 --- /dev/null +++ b/api/ml.rl.evaluation.html @@ -0,0 +1,314 @@ + + + + + + + ml.rl.evaluation package — ReAgent 1.0 documentation + + + + + + + + + + + + + + + + +
+ + +
+ +
+
+
+ +
+
+
+
+ +
+

ml.rl.evaluation package

+
+

Submodules

+
+
+

ml.rl.evaluation.cpe module

+
+
+class ml.rl.evaluation.cpe.CpeDetails
+

Bases: object

+
+
+log()
+
+ +
+
+log_to_tensorboard() → None
+
+ +
+ +
+
+class ml.rl.evaluation.cpe.CpeEstimate(raw, normalized, raw_std_error, normalized_std_error)
+

Bases: NamedTuple

+
+
+normalized: float
+

Alias for field number 1

+
+ +
+
+normalized_std_error: float
+

Alias for field number 3

+
+ +
+
+raw: float
+

Alias for field number 0

+
+ +
+
+raw_std_error: float
+

Alias for field number 2

+
+ +
+ +
+
+class ml.rl.evaluation.cpe.CpeEstimateSet(direct_method, inverse_propensity, doubly_robust, sequential_doubly_robust, weighted_doubly_robust, magic)
+

Bases: NamedTuple

+
+
+check_estimates_exist()
+
+ +
+
+direct_method: Optional[ml.rl.evaluation.cpe.CpeEstimate]
+

Alias for field number 0

+
+ +
+
+doubly_robust: Optional[ml.rl.evaluation.cpe.CpeEstimate]
+

Alias for field number 2

+
+ +
+
+fill_empty_with_zero()
+
+ +
+
+inverse_propensity: Optional[ml.rl.evaluation.cpe.CpeEstimate]
+

Alias for field number 1

+
+ +
+
+log()
+
+ +
+
+log_to_tensorboard(metric_name: str) → None
+
+ +
+
+magic: Optional[ml.rl.evaluation.cpe.CpeEstimate]
+

Alias for field number 5

+
+ +
+
+sequential_doubly_robust: Optional[ml.rl.evaluation.cpe.CpeEstimate]
+

Alias for field number 3

+
+ +
+
+weighted_doubly_robust: Optional[ml.rl.evaluation.cpe.CpeEstimate]
+

Alias for field number 4

+
+ +
+ +
+
+ml.rl.evaluation.cpe.bootstrapped_std_error_of_mean(data, sample_percent=0.25, num_samples=1000)
+

Compute bootstrapped standard error of mean of input data.

+
+
Parameters
+
    +
  • data – Input data (1D torch tensor or numpy array).

  • +
  • sample_percent – Size of sample to use to calculate bootstrap statistic.

  • +
  • num_samples – Number of times to sample.

  • +
+
+
+
+ +
+
+

ml.rl.evaluation.doubly_robust_estimator module

+
+
+

ml.rl.evaluation.evaluation_data_page module

+
+
+

ml.rl.evaluation.evaluator module

+
+
+

ml.rl.evaluation.ranking_evaluator module

+
+
+

ml.rl.evaluation.reward_net_evaluator module

+
+
+class ml.rl.evaluation.reward_net_evaluator.RewardNetEvaluator(trainer: ml.rl.training.reward_network_trainer.RewardNetTrainer)
+

Bases: object

+

Evaluate reward networks

+
+
+evaluate(eval_tdp: ml.rl.types.PreprocessedTrainingBatch)
+
+ +
+
+evaluate_post_training()
+
+ +
+ +
+
+

ml.rl.evaluation.sequential_doubly_robust_estimator module

+
+
+

ml.rl.evaluation.weighted_sequential_doubly_robust_estimator module

+
+
+

ml.rl.evaluation.world_model_evaluator module

+
+
+

Module contents

+
+
+ + +
+
+ +
+
+
+
+ + + + \ No newline at end of file diff --git a/api/ml.rl.html b/api/ml.rl.html new file mode 100644 index 00000000..92a0790d --- /dev/null +++ b/api/ml.rl.html @@ -0,0 +1,1404 @@ + + + + + + + ml.rl package — ReAgent 1.0 documentation + + + + + + + + + + + + + + + + +
+ + +
+ +
+
+
+ +
+
+
+
+ +
+

ml.rl package

+
+

Subpackages

+
+ +
+
+
+

Submodules

+
+
+

ml.rl.caffe_utils module

+
+
+class ml.rl.caffe_utils.C2
+

Bases: object

+
+
+static NextBlob(prefix: str) → str
+
+ +
+
+static init_net()
+
+ +
+
+static model()
+
+ +
+
+static net()
+
+ +
+
+static set_model(model)
+
+ +
+
+static set_net(net)
+
+ +
+
+static set_net_and_init_net(net, init_net)
+
+ +
+ +
+
+class ml.rl.caffe_utils.C2Meta
+

Bases: type

+
+ +
+
+

ml.rl.debug_on_error module

+
+
+ml.rl.debug_on_error.start()
+
+ +
+
+

ml.rl.json_serialize module

+
+
+ml.rl.json_serialize.from_json(j_obj: Any, to_type: Type) → Any
+
+ +
+
+ml.rl.json_serialize.isinstance_namedtuple(x)
+
+ +
+
+ml.rl.json_serialize.json_to_object(j: str, to_type: Type) → Any
+
+ +
+
+ml.rl.json_serialize.object_to_json(o: Any) → str
+
+ +
+
+ml.rl.json_serialize.prepare_for_json(o: Any) → Any
+
+ +
+
+

ml.rl.parameters module

+
+
+

ml.rl.tensorboardX module

+

Context library to allow dropping tensorboardX anywhere in the codebase. +If there is no SummaryWriter in the context, function calls will be no-op.

+

Usage:

+
+

writer = SummaryWriter()

+
+
with summary_writer_context(writer):

some_func()

+
+
def some_func():

SummaryWriterContext.add_scalar(“foo”, tensor)

+
+
+
+
+
+class ml.rl.tensorboardX.SummaryWriterContext
+

Bases: object

+
+
+classmethod add_custom_scalars(writer)
+

Call this once you are satisfied setting up custom scalar

+
+ +
+
+classmethod add_custom_scalars_multilinechart(tags, category=None, title=None)
+
+ +
+
+classmethod add_histogram(key, val, *args, **kwargs)
+
+ +
+
+classmethod increase_global_step()
+
+ +
+
+classmethod pop()
+
+ +
+
+classmethod push(writer)
+
+ +
+ +
+
+class ml.rl.tensorboardX.SummaryWriterContextMeta
+

Bases: type

+
+ +
+
+ml.rl.tensorboardX.summary_writer_context(writer)
+
+ +
+
+

ml.rl.torch_utils module

+
+
+ml.rl.torch_utils.export_module_to_buffer(module) → _io.BytesIO
+
+ +
+
+ml.rl.torch_utils.masked_softmax(x, mask, temperature)
+

Compute softmax values for each sets of scores in x.

+
+ +
+
+ml.rl.torch_utils.rescale_torch_tensor(tensor: torch.Tensor, new_min: torch.Tensor, new_max: torch.Tensor, prev_min: torch.Tensor, prev_max: torch.Tensor)
+

Rescale column values in N X M torch tensor to be in new range. +Each column m in input tensor will be rescaled from range +[prev_min[m], prev_max[m]] to [new_min[m], new_max[m]]

+
+ +
+
+ml.rl.torch_utils.softmax(x, temperature)
+

Compute softmax values for each sets of scores in x.

+
+ +
+
+ml.rl.torch_utils.stack(mems)
+

Stack a list of tensors +Could use torch.stack here but torch.stack is much slower +than torch.cat + view +Submitted an issue for investigation: +https://github.com/pytorch/pytorch/issues/22462

+

FIXME: Remove this function after the issue above is resolved

+
+ +
+
+

ml.rl.types module

+
+
+class ml.rl.types.ActorOutput(action: torch.Tensor, log_prob: Optional[torch.Tensor] = None, action_mean: Optional[torch.Tensor] = None)
+

Bases: ml.rl.types.BaseDataClass

+
+
+action: torch.Tensor
+
+ +
+
+action_mean: Optional[torch.Tensor] = None
+
+ +
+
+log_prob: Optional[torch.Tensor] = None
+
+ +
+ +
+
+class ml.rl.types.AllActionQValues(q_values: torch.Tensor)
+

Bases: ml.rl.types.BaseDataClass

+
+
+q_values: torch.Tensor
+
+ +
+ +
+
+class ml.rl.types.BaseDataClass
+

Bases: object

+
+
+cuda()
+
+ +
+
+pin_memory()
+
+ +
+ +
+
+class ml.rl.types.CommonInput(reward: torch.Tensor, time_diff: torch.Tensor, step: Optional[torch.Tensor], not_terminal: torch.Tensor)
+

Bases: ml.rl.types.BaseDataClass

+

Base class for all inputs, both raw and preprocessed

+
+
+not_terminal: torch.Tensor
+
+ +
+
+reward: torch.Tensor
+
+ +
+
+step: Optional[torch.Tensor]
+
+ +
+
+time_diff: torch.Tensor
+
+ +
+ +
+
+class ml.rl.types.DqnPolicyActionSet(greedy: int, softmax: Optional[int] = None, greedy_act_name: Optional[str] = None, softmax_act_name: Optional[str] = None, softmax_act_prob: Optional[float] = None)
+

Bases: ml.rl.types.BaseDataClass

+
+
+greedy: int
+
+ +
+
+greedy_act_name: Optional[str] = None
+
+ +
+
+softmax: Optional[int] = None
+
+ +
+
+softmax_act_name: Optional[str] = None
+
+ +
+
+softmax_act_prob: Optional[float] = None
+
+ +
+ +
+
+class ml.rl.types.ExtraData(mdp_id: Optional[numpy.ndarray] = None, sequence_number: Optional[torch.Tensor] = None, action_probability: Optional[torch.Tensor] = None, max_num_actions: Optional[int] = None, metrics: Optional[torch.Tensor] = None)
+

Bases: ml.rl.types.BaseDataClass

+
+
+action_probability: Optional[torch.Tensor] = None
+
+ +
+
+max_num_actions: Optional[int] = None
+
+ +
+
+mdp_id: Optional[numpy.ndarray] = None
+
+ +
+
+metrics: Optional[torch.Tensor] = None
+
+ +
+
+sequence_number: Optional[torch.Tensor] = None
+
+ +
+ +
+
+class ml.rl.types.FeatureVector(float_features: ml.rl.types.ValuePresence, id_list_features: Dict[str, Tuple[torch.Tensor, torch.Tensor]] = <factory>, time_since_first: Optional[torch.Tensor] = None)
+

Bases: ml.rl.types.BaseDataClass

+
+
+float_features: ml.rl.types.ValuePresence
+
+ +
+
+id_list_features: Dict[str, Tuple[torch.Tensor, torch.Tensor]]
+
+ +
+
+time_since_first: Optional[torch.Tensor] = None
+
+ +
+ +
+
+class ml.rl.types.FloatFeatureInfo(name: str, feature_id: int)
+

Bases: ml.rl.types.BaseDataClass

+
+
+feature_id: int
+
+ +
+
+name: str
+
+ +
+ +
+
+class ml.rl.types.IdListFeatureConfig(name: str, feature_id: int, id_mapping_name: str)
+

Bases: ml.rl.types.BaseDataClass

+

This describes how to map raw features to model features

+
+
+feature_id: int
+
+ +
+
+id_mapping_name: str
+
+ +
+
+name: str
+
+ +
+ +
+
+class ml.rl.types.IdMapping(ids: List[int])
+

Bases: ml.rl.types.BaseDataClass

+
+
+ids: List[int]
+
+ +
+ +
+
+class ml.rl.types.MemoryNetworkOutput(mus: torch.Tensor, sigmas: torch.Tensor, logpi: torch.Tensor, reward: torch.Tensor, not_terminal: torch.Tensor, last_step_lstm_hidden: torch.Tensor, last_step_lstm_cell: torch.Tensor, all_steps_lstm_hidden: torch.Tensor)
+

Bases: ml.rl.types.BaseDataClass

+
+
+all_steps_lstm_hidden: torch.Tensor
+
+ +
+
+last_step_lstm_cell: torch.Tensor
+
+ +
+
+last_step_lstm_hidden: torch.Tensor
+
+ +
+
+logpi: torch.Tensor
+
+ +
+
+mus: torch.Tensor
+
+ +
+
+not_terminal: torch.Tensor
+
+ +
+
+reward: torch.Tensor
+
+ +
+
+sigmas: torch.Tensor
+
+ +
+ +
+
+class ml.rl.types.ModelFeatureConfig(float_feature_infos: List[ml.rl.types.FloatFeatureInfo], id_mapping_config: Dict[str, ml.rl.types.IdMapping] = <factory>, id_list_feature_configs: List[ml.rl.types.IdListFeatureConfig] = <factory>)
+

Bases: ml.rl.types.BaseDataClass

+
+
+float_feature_infos: List[ml.rl.types.FloatFeatureInfo]
+
+ +
+
+id_list_feature_configs: List[ml.rl.types.IdListFeatureConfig]
+
+ +
+
+id_mapping_config: Dict[str, ml.rl.types.IdMapping]
+
+ +
+ +
+
+class ml.rl.types.PlanningPolicyOutput(next_best_continuous_action: Optional[torch.Tensor] = None, next_best_discrete_action_one_hot: Optional[torch.Tensor] = None, next_best_discrete_action_idx: Optional[int] = None)
+

Bases: ml.rl.types.BaseDataClass

+
+
+next_best_continuous_action: Optional[torch.Tensor] = None
+
+ +
+
+next_best_discrete_action_idx: Optional[int] = None
+
+ +
+
+next_best_discrete_action_one_hot: Optional[torch.Tensor] = None
+
+ +
+ +
+
+class ml.rl.types.PreprocessedBaseInput(reward: torch.Tensor, time_diff: torch.Tensor, step: Optional[torch.Tensor], not_terminal: torch.Tensor, state: ml.rl.types.PreprocessedFeatureVector, next_state: ml.rl.types.PreprocessedFeatureVector)
+

Bases: ml.rl.types.CommonInput

+
+
+next_state: ml.rl.types.PreprocessedFeatureVector
+
+ +
+
+state: ml.rl.types.PreprocessedFeatureVector
+
+ +
+ +
+
+class ml.rl.types.PreprocessedDiscreteDqnInput(reward: torch.Tensor, time_diff: torch.Tensor, step: Optional[torch.Tensor], not_terminal: torch.Tensor, state: ml.rl.types.PreprocessedFeatureVector, next_state: ml.rl.types.PreprocessedFeatureVector, action: torch.Tensor, next_action: torch.Tensor, possible_actions_mask: torch.Tensor, possible_next_actions_mask: torch.Tensor)
+

Bases: ml.rl.types.PreprocessedBaseInput

+
+
+action: torch.Tensor
+
+ +
+
+next_action: torch.Tensor
+
+ +
+
+possible_actions_mask: torch.Tensor
+
+ +
+
+possible_next_actions_mask: torch.Tensor
+
+ +
+ +
+
+class ml.rl.types.PreprocessedFeatureVector(float_features: torch.Tensor, id_list_features: Dict[str, Tuple[torch.Tensor, torch.Tensor]] = <factory>, time_since_first: Optional[torch.Tensor] = None)
+

Bases: ml.rl.types.BaseDataClass

+
+
+float_features: torch.Tensor
+
+ +
+
+id_list_features: Dict[str, Tuple[torch.Tensor, torch.Tensor]]
+
+ +
+
+time_since_first: Optional[torch.Tensor] = None
+
+ +
+ +
+
+class ml.rl.types.PreprocessedMemoryNetworkInput(reward: torch.Tensor, time_diff: torch.Tensor, step: Optional[torch.Tensor], not_terminal: torch.Tensor, state: ml.rl.types.PreprocessedFeatureVector, next_state: ml.rl.types.PreprocessedFeatureVector, action: Union[torch.Tensor, torch.Tensor])
+

Bases: ml.rl.types.PreprocessedBaseInput

+
+
+action: Union[torch.Tensor, torch.Tensor]
+
+ +
+ +
+
+class ml.rl.types.PreprocessedParametricDqnInput(reward: torch.Tensor, time_diff: torch.Tensor, step: Optional[torch.Tensor], not_terminal: torch.Tensor, state: ml.rl.types.PreprocessedFeatureVector, next_state: ml.rl.types.PreprocessedFeatureVector, action: ml.rl.types.PreprocessedFeatureVector, next_action: ml.rl.types.PreprocessedFeatureVector, possible_actions: ml.rl.types.PreprocessedFeatureVector, possible_actions_mask: torch.Tensor, possible_next_actions: ml.rl.types.PreprocessedFeatureVector, possible_next_actions_mask: torch.Tensor, tiled_next_state: ml.rl.types.PreprocessedFeatureVector)
+

Bases: ml.rl.types.PreprocessedBaseInput

+
+
+action: ml.rl.types.PreprocessedFeatureVector
+
+ +
+
+next_action: ml.rl.types.PreprocessedFeatureVector
+
+ +
+
+possible_actions: ml.rl.types.PreprocessedFeatureVector
+
+ +
+
+possible_actions_mask: torch.Tensor
+
+ +
+
+possible_next_actions: ml.rl.types.PreprocessedFeatureVector
+
+ +
+
+possible_next_actions_mask: torch.Tensor
+
+ +
+
+tiled_next_state: ml.rl.types.PreprocessedFeatureVector
+
+ +
+ +
+
+class ml.rl.types.PreprocessedPolicyNetworkInput(reward: torch.Tensor, time_diff: torch.Tensor, step: Optional[torch.Tensor], not_terminal: torch.Tensor, state: ml.rl.types.PreprocessedFeatureVector, next_state: ml.rl.types.PreprocessedFeatureVector, action: ml.rl.types.PreprocessedFeatureVector, next_action: ml.rl.types.PreprocessedFeatureVector)
+

Bases: ml.rl.types.PreprocessedBaseInput

+
+
+action: ml.rl.types.PreprocessedFeatureVector
+
+ +
+
+next_action: ml.rl.types.PreprocessedFeatureVector
+
+ +
+ +
+
+class ml.rl.types.PreprocessedRankingInput(state: ml.rl.types.PreprocessedFeatureVector, src_seq: ml.rl.types.PreprocessedFeatureVector, src_src_mask: torch.Tensor, tgt_in_seq: Optional[ml.rl.types.PreprocessedFeatureVector] = None, tgt_out_seq: Optional[ml.rl.types.PreprocessedFeatureVector] = None, tgt_tgt_mask: Optional[torch.Tensor] = None, slate_reward: Optional[torch.Tensor] = None, src_in_idx: Optional[torch.Tensor] = None, tgt_in_idx: Optional[torch.Tensor] = None, tgt_out_idx: Optional[torch.Tensor] = None, tgt_out_probs: Optional[torch.Tensor] = None, optim_tgt_in_idx: Optional[torch.Tensor] = None, optim_tgt_out_idx: Optional[torch.Tensor] = None, optim_tgt_in_seq: Optional[ml.rl.types.PreprocessedFeatureVector] = None, optim_tgt_out_seq: Optional[ml.rl.types.PreprocessedFeatureVector] = None)
+

Bases: ml.rl.types.BaseDataClass

+
+
+classmethod from_tensors(state: torch.Tensor, src_seq: torch.Tensor, src_src_mask: torch.Tensor, tgt_in_seq: Optional[torch.Tensor] = None, tgt_out_seq: Optional[torch.Tensor] = None, tgt_tgt_mask: Optional[torch.Tensor] = None, slate_reward: Optional[torch.Tensor] = None, src_in_idx: Optional[torch.Tensor] = None, tgt_in_idx: Optional[torch.Tensor] = None, tgt_out_idx: Optional[torch.Tensor] = None, tgt_out_probs: Optional[torch.Tensor] = None, optim_tgt_in_idx: Optional[torch.Tensor] = None, optim_tgt_out_idx: Optional[torch.Tensor] = None, optim_tgt_in_seq: Optional[torch.Tensor] = None, optim_tgt_out_seq: Optional[torch.Tensor] = None)
+
+ +
+
+optim_tgt_in_idx: Optional[torch.Tensor] = None
+
+ +
+
+optim_tgt_in_seq: Optional[ml.rl.types.PreprocessedFeatureVector] = None
+
+ +
+
+optim_tgt_out_idx: Optional[torch.Tensor] = None
+
+ +
+
+optim_tgt_out_seq: Optional[ml.rl.types.PreprocessedFeatureVector] = None
+
+ +
+
+slate_reward: Optional[torch.Tensor] = None
+
+ +
+
+src_in_idx: Optional[torch.Tensor] = None
+
+ +
+
+src_seq: ml.rl.types.PreprocessedFeatureVector
+
+ +
+
+src_src_mask: torch.Tensor
+
+ +
+
+state: ml.rl.types.PreprocessedFeatureVector
+
+ +
+
+tgt_in_idx: Optional[torch.Tensor] = None
+
+ +
+
+tgt_in_seq: Optional[ml.rl.types.PreprocessedFeatureVector] = None
+
+ +
+
+tgt_out_idx: Optional[torch.Tensor] = None
+
+ +
+
+tgt_out_probs: Optional[torch.Tensor] = None
+
+ +
+
+tgt_out_seq: Optional[ml.rl.types.PreprocessedFeatureVector] = None
+
+ +
+
+tgt_tgt_mask: Optional[torch.Tensor] = None
+
+ +
+ +
+
+class ml.rl.types.PreprocessedSlateFeatureVector(float_features: torch.Tensor, item_mask: torch.Tensor, item_probability: torch.Tensor)
+

Bases: ml.rl.types.BaseDataClass

+

The shape of float_features is +(batch_size, slate_size, item_dim).

+

item_mask masks available items in the action

+

item_probability is the probability of item in being selected

+
+
+as_preprocessed_feature_vector() → ml.rl.types.PreprocessedFeatureVector
+
+ +
+
+float_features: torch.Tensor
+
+ +
+
+item_mask: torch.Tensor
+
+ +
+
+item_probability: torch.Tensor
+
+ +
+ +
+
+class ml.rl.types.PreprocessedSlateQInput(reward: torch.Tensor, time_diff: torch.Tensor, step: Optional[torch.Tensor], not_terminal: torch.Tensor, state: ml.rl.types.PreprocessedFeatureVector, next_state: ml.rl.types.PreprocessedFeatureVector, tiled_state: ml.rl.types.PreprocessedTiledFeatureVector, tiled_next_state: ml.rl.types.PreprocessedTiledFeatureVector, action: ml.rl.types.PreprocessedSlateFeatureVector, next_action: ml.rl.types.PreprocessedSlateFeatureVector, reward_mask: torch.Tensor)
+

Bases: ml.rl.types.PreprocessedBaseInput

+

The shapes of tiled_state & tiled_next_state are +(batch_size, slate_size, state_dim).

+

The shapes of reward, reward_mask, & next_item_mask are +(batch_size, slate_size).

+

reward_mask indicated whether the reward could be observed, e.g., +the item got into viewport or not.

+
+
+action: ml.rl.types.PreprocessedSlateFeatureVector
+
+ +
+
+next_action: ml.rl.types.PreprocessedSlateFeatureVector
+
+ +
+
+reward_mask: torch.Tensor
+
+ +
+
+tiled_next_state: ml.rl.types.PreprocessedTiledFeatureVector
+
+ +
+
+tiled_state: ml.rl.types.PreprocessedTiledFeatureVector
+
+ +
+ +
+
+class ml.rl.types.PreprocessedState(state)
+

Bases: ml.rl.types.BaseDataClass

+

This class makes it easier to plug modules into predictor

+
+
+classmethod from_tensor(state: torch.Tensor)
+
+ +
+
+state: ml.rl.types.PreprocessedFeatureVector
+
+ +
+ +
+
+class ml.rl.types.PreprocessedStateAction(state, action)
+

Bases: ml.rl.types.BaseDataClass

+
+
+action: ml.rl.types.PreprocessedFeatureVector
+
+ +
+
+classmethod from_tensors(state: torch.Tensor, action: torch.Tensor)
+
+ +
+
+state: ml.rl.types.PreprocessedFeatureVector
+
+ +
+ +
+
+class ml.rl.types.PreprocessedTiledFeatureVector(float_features: torch.Tensor)
+

Bases: ml.rl.types.BaseDataClass

+
+
+as_preprocessed_feature_vector() → ml.rl.types.PreprocessedFeatureVector
+
+ +
+
+float_features: torch.Tensor
+
+ +
+ +
+
+class ml.rl.types.PreprocessedTrainingBatch(training_input: Union[ml.rl.types.PreprocessedBaseInput, ml.rl.types.PreprocessedDiscreteDqnInput, ml.rl.types.PreprocessedParametricDqnInput, ml.rl.types.PreprocessedMemoryNetworkInput, ml.rl.types.PreprocessedPolicyNetworkInput, ml.rl.types.PreprocessedRankingInput, ml.rl.types.PreprocessedSlateQInput], extras: Any)
+

Bases: ml.rl.types.BaseDataClass

+
+
+batch_size()
+
+ +
+
+extras: Any
+
+ +
+
+training_input: Union[ml.rl.types.PreprocessedBaseInput, ml.rl.types.PreprocessedDiscreteDqnInput, ml.rl.types.PreprocessedParametricDqnInput, ml.rl.types.PreprocessedMemoryNetworkInput, ml.rl.types.PreprocessedPolicyNetworkInput, ml.rl.types.PreprocessedRankingInput, ml.rl.types.PreprocessedSlateQInput]
+
+ +
+ +
+
+class ml.rl.types.RankingOutput(ranked_tgt_out_idx: Optional[torch.Tensor] = None, ranked_tgt_out_probs: Optional[torch.Tensor] = None, log_probs: Optional[torch.Tensor] = None)
+

Bases: ml.rl.types.BaseDataClass

+
+
+log_probs: Optional[torch.Tensor] = None
+
+ +
+
+ranked_tgt_out_idx: Optional[torch.Tensor] = None
+
+ +
+
+ranked_tgt_out_probs: Optional[torch.Tensor] = None
+
+ +
+ +
+
+class ml.rl.types.RawBaseInput(reward: torch.Tensor, time_diff: torch.Tensor, step: Optional[torch.Tensor], not_terminal: torch.Tensor, state: ml.rl.types.FeatureVector, next_state: ml.rl.types.FeatureVector)
+

Bases: ml.rl.types.CommonInput

+
+
+next_state: ml.rl.types.FeatureVector
+
+ +
+
+state: ml.rl.types.FeatureVector
+
+ +
+ +
+
+class ml.rl.types.RawDiscreteDqnInput(reward: torch.Tensor, time_diff: torch.Tensor, step: Optional[torch.Tensor], not_terminal: torch.Tensor, state: ml.rl.types.FeatureVector, next_state: ml.rl.types.FeatureVector, action: torch.Tensor, next_action: torch.Tensor, possible_actions_mask: torch.Tensor, possible_next_actions_mask: torch.Tensor)
+

Bases: ml.rl.types.RawBaseInput

+
+
+action: torch.Tensor
+
+ +
+
+next_action: torch.Tensor
+
+ +
+
+possible_actions_mask: torch.Tensor
+
+ +
+
+possible_next_actions_mask: torch.Tensor
+
+ +
+
+preprocess(state: ml.rl.types.PreprocessedFeatureVector, next_state: ml.rl.types.PreprocessedFeatureVector)
+
+ +
+
+preprocess_tensors(state: torch.Tensor, next_state: torch.Tensor)
+
+ +
+ +
+
+class ml.rl.types.RawMemoryNetworkInput(reward: torch.Tensor, time_diff: torch.Tensor, step: Optional[torch.Tensor], not_terminal: torch.Tensor, state: ml.rl.types.FeatureVector, next_state: ml.rl.types.FeatureVector, action: Union[ml.rl.types.FeatureVector, torch.Tensor])
+

Bases: ml.rl.types.RawBaseInput

+
+
+action: Union[ml.rl.types.FeatureVector, torch.Tensor]
+
+ +
+
+preprocess(state: ml.rl.types.PreprocessedFeatureVector, next_state: ml.rl.types.PreprocessedFeatureVector, action: Optional[torch.Tensor] = None)
+
+ +
+
+preprocess_tensors(state: torch.Tensor, next_state: torch.Tensor, action: Optional[torch.Tensor] = None)
+
+ +
+ +
+
+class ml.rl.types.RawParametricDqnInput(reward: torch.Tensor, time_diff: torch.Tensor, step: Optional[torch.Tensor], not_terminal: torch.Tensor, state: ml.rl.types.FeatureVector, next_state: ml.rl.types.FeatureVector, action: ml.rl.types.FeatureVector, next_action: ml.rl.types.FeatureVector, possible_actions: ml.rl.types.FeatureVector, possible_actions_mask: torch.Tensor, possible_next_actions: ml.rl.types.FeatureVector, possible_next_actions_mask: torch.Tensor, tiled_next_state: ml.rl.types.FeatureVector)
+

Bases: ml.rl.types.RawBaseInput

+
+
+action: ml.rl.types.FeatureVector
+
+ +
+
+next_action: ml.rl.types.FeatureVector
+
+ +
+
+possible_actions: ml.rl.types.FeatureVector
+
+ +
+
+possible_actions_mask: torch.Tensor
+
+ +
+
+possible_next_actions: ml.rl.types.FeatureVector
+
+ +
+
+possible_next_actions_mask: torch.Tensor
+
+ +
+
+preprocess(state: ml.rl.types.PreprocessedFeatureVector, next_state: ml.rl.types.PreprocessedFeatureVector, action: ml.rl.types.PreprocessedFeatureVector, next_action: ml.rl.types.PreprocessedFeatureVector, possible_actions: ml.rl.types.PreprocessedFeatureVector, possible_next_actions: ml.rl.types.PreprocessedFeatureVector, tiled_next_state: ml.rl.types.PreprocessedFeatureVector)
+
+ +
+
+preprocess_tensors(state: torch.Tensor, next_state: torch.Tensor, action: torch.Tensor, next_action: torch.Tensor, possible_actions: torch.Tensor, possible_next_actions: torch.Tensor, tiled_next_state: torch.Tensor)
+
+ +
+
+tiled_next_state: ml.rl.types.FeatureVector
+
+ +
+ +
+
+class ml.rl.types.RawPolicyNetworkInput(reward: torch.Tensor, time_diff: torch.Tensor, step: Optional[torch.Tensor], not_terminal: torch.Tensor, state: ml.rl.types.FeatureVector, next_state: ml.rl.types.FeatureVector, action: ml.rl.types.FeatureVector, next_action: ml.rl.types.FeatureVector)
+

Bases: ml.rl.types.RawBaseInput

+
+
+action: ml.rl.types.FeatureVector
+
+ +
+
+next_action: ml.rl.types.FeatureVector
+
+ +
+
+preprocess(state: ml.rl.types.PreprocessedFeatureVector, next_state: ml.rl.types.PreprocessedFeatureVector, action: ml.rl.types.PreprocessedFeatureVector, next_action: ml.rl.types.PreprocessedFeatureVector)
+
+ +
+
+preprocess_tensors(state: torch.Tensor, next_state: torch.Tensor, action: torch.Tensor, next_action: torch.Tensor)
+
+ +
+ +
+
+class ml.rl.types.RawStateAction(state: ml.rl.types.FeatureVector, action: ml.rl.types.FeatureVector)
+

Bases: ml.rl.types.BaseDataClass

+
+
+action: ml.rl.types.FeatureVector
+
+ +
+
+state: ml.rl.types.FeatureVector
+
+ +
+ +
+
+class ml.rl.types.RawTrainingBatch(training_input: Union[ml.rl.types.RawBaseInput, ml.rl.types.RawDiscreteDqnInput, ml.rl.types.RawParametricDqnInput, ml.rl.types.RawPolicyNetworkInput], extras: Any)
+

Bases: ml.rl.types.BaseDataClass

+
+
+batch_size()
+
+ +
+
+extras: Any
+
+ +
+
+preprocess(training_input: Union[ml.rl.types.PreprocessedBaseInput, ml.rl.types.PreprocessedDiscreteDqnInput, ml.rl.types.PreprocessedParametricDqnInput, ml.rl.types.PreprocessedMemoryNetworkInput, ml.rl.types.PreprocessedPolicyNetworkInput]) → ml.rl.types.PreprocessedTrainingBatch
+
+ +
+
+training_input: Union[ml.rl.types.RawBaseInput, ml.rl.types.RawDiscreteDqnInput, ml.rl.types.RawParametricDqnInput, ml.rl.types.RawPolicyNetworkInput]
+
+ +
+ +
+
+class ml.rl.types.RewardNetworkOutput(predicted_reward: torch.Tensor)
+

Bases: ml.rl.types.BaseDataClass

+
+
+predicted_reward: torch.Tensor
+
+ +
+ +
+
+class ml.rl.types.SacPolicyActionSet(greedy: torch.Tensor, greedy_propensity: float)
+

Bases: ml.rl.types.BaseDataClass

+
+
+greedy: torch.Tensor
+
+ +
+
+greedy_propensity: float
+
+ +
+ +
+
+class ml.rl.types.SingleQValue(q_value: torch.Tensor)
+

Bases: ml.rl.types.BaseDataClass

+
+
+q_value: torch.Tensor
+
+ +
+ +
+
+class ml.rl.types.ValuePresence(value: torch.Tensor, presence: Optional[torch.Tensor])
+

Bases: ml.rl.types.BaseDataClass

+
+
+presence: Optional[torch.Tensor]
+
+ +
+
+value: torch.Tensor
+
+ +
+ +
+
+

Module contents

+
+
+ + +
+
+ +
+
+
+
+ + + + \ No newline at end of file diff --git a/api/ml.rl.models.html b/api/ml.rl.models.html new file mode 100644 index 00000000..3e879ea7 --- /dev/null +++ b/api/ml.rl.models.html @@ -0,0 +1,1246 @@ + + + + + + + ml.rl.models package — ReAgent 1.0 documentation + + + + + + + + + + + + + + + + +
+ + +
+ +
+
+
+ +
+
+
+
+ +
+

ml.rl.models package

+
+

Submodules

+
+
+

ml.rl.models.actor module

+
+
+class ml.rl.models.actor.DirichletFullyConnectedActor(*args: Any, **kwargs: Any)
+

Bases: ml.rl.models.base.ModelBase

+
+
+EPSILON = 1e-06
+
+ +
+
+forward(input)
+
+ +
+
+get_log_prob(state, action)
+
+ +
+
+input_prototype()
+

This function provides the input for ONNX graph tracing.

+

The return value should be what expected by forward().

+
+ +
+ +
+
+class ml.rl.models.actor.FullyConnectedActor(*args: Any, **kwargs: Any)
+

Bases: ml.rl.models.base.ModelBase

+
+
+forward(input)
+
+ +
+
+input_prototype()
+

This function provides the input for ONNX graph tracing.

+

The return value should be what expected by forward().

+
+ +
+ +
+
+class ml.rl.models.actor.GaussianFullyConnectedActor(*args: Any, **kwargs: Any)
+

Bases: ml.rl.models.base.ModelBase

+
+
+forward(input)
+
+ +
+
+get_log_prob(state, squashed_action)
+

Action is expected to be squashed with tanh

+
+ +
+
+input_prototype()
+

This function provides the input for ONNX graph tracing.

+

The return value should be what expected by forward().

+
+ +
+ +
+
+

ml.rl.models.base module

+
+
+class ml.rl.models.base.ModelBase(*args: Any, **kwargs: Any)
+

Bases: torch.nn.Module

+

A base class to support exporting through ONNX

+
+
+cpu_model()
+

Override this in DistributedDataParallel models

+
+ +
+
+feature_config() → Optional[ml.rl.types.ModelFeatureConfig]
+

If the model needs additional preprocessing, e.g., using sequence features, +returns the config here.

+
+ +
+
+get_distributed_data_parallel_model()
+

Return DistributedDataParallel version of this model

+

This needs to be implemented explicitly because: +1) Model with EmbeddingBag module is not compatible with vanilla DistributedDataParallel +2) Exporting logic needs structured data. DistributedDataParallel doesn’t work with structured data.

+
+ +
+
+get_target_network()
+

Return a copy of this network to be used as target network

+

Subclass should override this if the target network should share parameters +with the network to be trained.

+
+ +
+
+input_prototype() → Any
+

This function provides the input for ONNX graph tracing.

+

The return value should be what expected by forward().

+
+ +
+ +
+
+

ml.rl.models.bcq module

+
+
+class ml.rl.models.bcq.BatchConstrainedDQN(*args: Any, **kwargs: Any)
+

Bases: ml.rl.models.base.ModelBase

+
+
+forward(input)
+
+ +
+
+input_prototype()
+

This function provides the input for ONNX graph tracing.

+

The return value should be what expected by forward().

+
+ +
+ +
+
+

ml.rl.models.categorical_dqn module

+
+
+class ml.rl.models.categorical_dqn.CategoricalDQN(*args: Any, **kwargs: Any)
+

Bases: ml.rl.models.base.ModelBase

+
+
+forward(input: ml.rl.types.PreprocessedState)
+
+ +
+
+input_prototype()
+

This function provides the input for ONNX graph tracing.

+

The return value should be what expected by forward().

+
+ +
+
+log_dist(input: ml.rl.types.PreprocessedState)
+
+ +
+
+serving_model()
+
+ +
+ +
+
+

ml.rl.models.cem_planner module

+

A network which implements a cross entropy method-based planner

+

The planner plans the best next action based on simulation data generated by +an ensemble of world models.

+

The idea is inspired by: https://arxiv.org/abs/1805.12114

+
+
+class ml.rl.models.cem_planner.CEMPlanner(*args: Any, **kwargs: Any)
+

Bases: ml.rl.models.base.ModelBase

+
+
+forward(input: ml.rl.types.PreprocessedState)
+
+ +
+
+get_distributed_data_parallel_model()
+

Return DistributedDataParallel version of this model

+

This needs to be implemented explicitly because: +1) Model with EmbeddingBag module is not compatible with vanilla DistributedDataParallel +2) Exporting logic needs structured data. DistributedDataParallel doesn’t work with structured data.

+
+ +
+
+input_prototype()
+

This function provides the input for ONNX graph tracing.

+

The return value should be what expected by forward().

+
+ +
+ +
+
+class ml.rl.models.cem_planner.CEMPlannerNetwork(mem_net_list: List[ml.rl.models.world_model.MemoryNetwork], cem_num_iterations: int, cem_population_size: int, ensemble_population_size: int, num_elites: int, plan_horizon_length: int, state_dim: int, action_dim: int, discrete_action: bool, terminal_effective: bool, gamma: float, alpha: float = 0.25, epsilon: float = 0.001, action_upper_bounds: Optional[numpy.ndarray] = None, action_lower_bounds: Optional[numpy.ndarray] = None)
+

Bases: torch.nn.modules.module.Module

+
+
+acc_rewards_of_all_solutions(input: ml.rl.types.PreprocessedState, solutions: torch.Tensor) → float
+

Calculate accumulated rewards of solutions.

+
+
Parameters
+
    +
  • input – the input which contains the starting state

  • +
  • solutions – its shape is (cem_pop_size, plan_horizon_length, action_dim)

  • +
+
+
Returns
+

a vector of size cem_pop_size, which is the reward of each solution

+
+
+
+ +
+
+acc_rewards_of_one_solution(init_state: torch.Tensor, solution: torch.Tensor, solution_idx: int)
+

ensemble_pop_size trajectories will be sampled to evaluate a +CEM solution. Each trajectory is generated by one world model

+
+
Parameters
+
    +
  • init_state – its shape is (state_dim, )

  • +
  • solution – its shape is (plan_horizon_length, action_dim)

  • +
  • solution_idx – the index of the solution

  • +
+
+
Return reward
+

Reward of each of ensemble_pop_size trajectories

+
+
+
+ +
+
+constrained_variance(mean, var)
+
+ +
+
+continuous_planning(input: ml.rl.types.PreprocessedState) → numpy.ndarray
+
+ +
+
+discrete_planning(input: ml.rl.types.PreprocessedState) → Tuple[int, numpy.ndarray]
+
+ +
+
+forward(input: ml.rl.types.PreprocessedState)
+

Defines the computation performed at every call.

+

Should be overridden by all subclasses.

+
+

Note

+

Although the recipe for forward pass needs to be defined within +this function, one should call the Module instance afterwards +instead of this since the former takes care of running the +registered hooks while the latter silently ignores them.

+
+
+ +
+
+sample_reward_next_state_terminal(world_model_input: ml.rl.types.PreprocessedStateAction, mem_net: ml.rl.models.world_model.MemoryNetwork)
+

Sample one-step dynamics based on the provided world model

+
+ +
+
+training: bool
+
+ +
+ +
+
+

ml.rl.models.convolutional_network module

+
+
+class ml.rl.models.convolutional_network.ConvolutionalNetwork(cnn_parameters, layers, activations)
+

Bases: torch.nn.modules.module.Module

+
+
+conv_forward(input)
+
+ +
+
+forward(input) → torch.FloatTensor
+

Forward pass for generic convnet DNNs. Assumes activation names +are valid pytorch activation names. +:param input image tensor

+
+ +
+
+training: bool
+
+ +
+ +
+
+

ml.rl.models.dqn module

+
+
+class ml.rl.models.dqn.FullyConnectedDQN(*args: Any, **kwargs: Any)
+

Bases: ml.rl.models.base.ModelBase

+
+
+forward(input: ml.rl.types.PreprocessedState)
+
+ +
+
+get_distributed_data_parallel_model()
+

Return DistributedDataParallel version of this model

+

This needs to be implemented explicitly because: +1) Model with EmbeddingBag module is not compatible with vanilla DistributedDataParallel +2) Exporting logic needs structured data. DistributedDataParallel doesn’t work with structured data.

+
+ +
+
+input_prototype()
+

This function provides the input for ONNX graph tracing.

+

The return value should be what expected by forward().

+
+ +
+ +
+
+

ml.rl.models.dueling_q_network module

+
+
+class ml.rl.models.dueling_q_network.DuelingQNetwork(*args: Any, **kwargs: Any)
+

Bases: ml.rl.models.base.ModelBase

+
+
+forward(input) → Union[NamedTuple, torch.FloatTensor]
+
+ +
+
+input_prototype()
+

This function provides the input for ONNX graph tracing.

+

The return value should be what expected by forward().

+
+ +
+ +
+
+

ml.rl.models.dueling_quantile_dqn module

+
+
+class ml.rl.models.dueling_quantile_dqn.DuelingQuantileDQN(*args: Any, **kwargs: Any)
+

Bases: ml.rl.models.base.ModelBase

+
+
+dist(input: ml.rl.types.PreprocessedState)
+
+ +
+
+forward(input: ml.rl.types.PreprocessedState)
+
+ +
+
+input_prototype()
+

This function provides the input for ONNX graph tracing.

+

The return value should be what expected by forward().

+
+ +
+ +
+
+

ml.rl.models.example_sequence_model module

+
+
+class ml.rl.models.example_sequence_model.ExampleSequenceModel(*args: Any, **kwargs: Any)
+

Bases: ml.rl.models.base.ModelBase

+
+
+feature_config()
+

If the model needs additional preprocessing, e.g., using sequence features, +returns the config here.

+
+ +
+
+forward(state)
+
+ +
+
+input_prototype()
+

This function provides the input for ONNX graph tracing.

+

The return value should be what expected by forward().

+
+ +
+ +
+
+class ml.rl.models.example_sequence_model.ExampleSequenceModelOutput(value: torch.Tensor)
+

Bases: object

+
+
+value: torch.Tensor
+
+ +
+ +
+
+

ml.rl.models.fully_connected_network module

+
+
+class ml.rl.models.fully_connected_network.FullyConnectedNetwork(*args: Any, **kwargs: Any)
+

Bases: torch.nn.Module

+
+
+forward(input: torch.FloatTensor) → torch.FloatTensor
+

Forward pass for generic feed-forward DNNs. Assumes activation names +are valid pytorch activation names. +:param input tensor

+
+ +
+ +
+
+ml.rl.models.fully_connected_network.gaussian_fill_w_gain(tensor, activation, dim_in, min_std=0.0) → None
+

Gaussian initialization with gain.

+
+ +
+
+

ml.rl.models.mdn_rnn module

+
+
+class ml.rl.models.mdn_rnn.MDNRNN(*args: Any, **kwargs: Any)
+

Bases: ml.rl.models.mdn_rnn._MDNRNNBase

+

Mixture Density Network - Recurrent Neural Network

+
+
+forward(actions, states, hidden=None)
+

Forward pass of MDN-RNN

+
+
Parameters
+
    +
  • actions – (SEQ_LEN, BATCH_SIZE, ACTION_DIM) torch tensor

  • +
  • states – (SEQ_LEN, BATCH_SIZE, STATE_DIM) torch tensor

  • +
+
+
Returns
+

parameters of the GMM prediction for the next state,

+
+
+

gaussian prediction of the reward and logit prediction of +non-terminality. And the RNN’s outputs.

+
+
    +
  • mus: (SEQ_LEN, BATCH_SIZE, NUM_GAUSSIANS, STATE_DIM) torch tensor

  • +
  • sigmas: (SEQ_LEN, BATCH_SIZE, NUM_GAUSSIANS, STATE_DIM) torch tensor

  • +
  • logpi: (SEQ_LEN, BATCH_SIZE, NUM_GAUSSIANS) torch tensor

  • +
  • reward: (SEQ_LEN, BATCH_SIZE) torch tensor

  • +
  • not_terminal: (SEQ_LEN, BATCH_SIZE) torch tensor

  • +
  • +
    last_step_hidden_and_cell: TUPLE(

    (NUM_LAYERS, BATCH_SIZE, HIDDEN_SIZE), +(NUM_LAYERS, BATCH_SIZE, HIDDEN_SIZE)

    +
    +
    +
  • +
+

) torch tensor +- all_steps_hidden: (SEQ_LEN, BATCH_SIZE, HIDDEN_SIZE) torch tensor

+
+
+ +
+
+get_initial_hidden_state(batch_size=1)
+
+ +
+ +
+
+class ml.rl.models.mdn_rnn.MDNRNNMemoryPool(max_replay_memory_size)
+

Bases: object

+
+
+deque_sample(indices)
+
+ +
+
+insert_into_memory(state, action, next_state, reward, not_terminal)
+
+ +
+
+property memory_size
+
+ +
+
+sample_memories(batch_size, use_gpu=False, batch_first=False) → ml.rl.types.PreprocessedTrainingBatch
+
+
Parameters
+
    +
  • batch_size – number of samples to return

  • +
  • use_gpu – whether to put samples on gpu

  • +
  • batch_first – If True, the first dimension of data is batch_size. +If False (default), the first dimension is SEQ_LEN. Therefore, +state’s shape is SEQ_LEN x BATCH_SIZE x STATE_DIM, for example. By default, +MDN-RNN consumes data with SEQ_LEN as the first dimension.

  • +
+
+
+
+ +
+ +
+
+class ml.rl.models.mdn_rnn.MDNRNNMemorySample(state, action, next_state, reward, not_terminal)
+

Bases: NamedTuple

+
+
+action: numpy.ndarray
+

Alias for field number 1

+
+ +
+
+next_state: numpy.ndarray
+

Alias for field number 2

+
+ +
+
+not_terminal: float
+

Alias for field number 4

+
+ +
+
+reward: float
+

Alias for field number 3

+
+ +
+
+state: numpy.ndarray
+

Alias for field number 0

+
+ +
+ +
+
+ml.rl.models.mdn_rnn.gmm_loss(batch, mus, sigmas, logpi, reduce=True)
+

Computes the gmm loss.

+

Compute minus the log probability of batch under the GMM model described +by mus, sigmas, pi. Precisely, with bs1, bs2, … the sizes of the batch +dimensions (several batch dimension are useful when you have both a batch +axis and a time step axis), gs the number of mixtures and fs the number of +features.

+
+
Parameters
+
    +
  • batch – (bs1, bs2, *, fs) torch tensor

  • +
  • mus – (bs1, bs2, *, gs, fs) torch tensor

  • +
  • sigmas – (bs1, bs2, *, gs, fs) torch tensor

  • +
  • logpi – (bs1, bs2, *, gs) torch tensor

  • +
  • reduce – if not reduce, the mean in the following formula is omitted

  • +
+
+
Returns
+

+
+
+
+
loss(batch) = - mean_{i1=0..bs1, i2=0..bs2, …} log(
+
sum_{k=1..gs} pi[i1, i2, …, k] * N(

batch[i1, i2, …, :] | mus[i1, i2, …, k, :], sigmas[i1, i2, …, k, :]))

+
+
+
+
+

NOTE: The loss is not reduced along the feature dimension (i.e. it should +scale linearily with fs).

+

Adapted from: https://github.com/ctallec/world-models

+
+ +
+
+ml.rl.models.mdn_rnn.transpose(*args)
+
+ +
+
+

ml.rl.models.no_soft_update_embedding module

+
+
+class ml.rl.models.no_soft_update_embedding.NoSoftUpdateEmbedding(num_embeddings: int, embedding_dim: int, padding_idx: Optional[int] = None, max_norm: Optional[float] = None, norm_type: float = 2.0, scale_grad_by_freq: bool = False, sparse: bool = False, _weight: Optional[torch.Tensor] = None, device=None, dtype=None)
+

Bases: torch.nn.modules.sparse.Embedding

+

Use this instead of vanilla Embedding module to avoid soft-updating the embedding +table in the target network.

+
+
+embedding_dim: int
+
+ +
+
+max_norm: Optional[float]
+
+ +
+
+norm_type: float
+
+ +
+
+num_embeddings: int
+
+ +
+
+padding_idx: Optional[int]
+
+ +
+
+scale_grad_by_freq: bool
+
+ +
+
+sparse: bool
+
+ +
+
+weight: torch.Tensor
+
+ +
+ +
+
+

ml.rl.models.parametric_dqn module

+
+
+class ml.rl.models.parametric_dqn.FullyConnectedParametricDQN(*args: Any, **kwargs: Any)
+

Bases: ml.rl.models.base.ModelBase

+
+
+forward(input)
+
+ +
+
+get_distributed_data_parallel_model()
+

Return DistributedDataParallel version of this model

+

This needs to be implemented explicitly because: +1) Model with EmbeddingBag module is not compatible with vanilla DistributedDataParallel +2) Exporting logic needs structured data. DistributedDataParallel doesn’t work with structured data.

+
+ +
+
+input_prototype()
+

This function provides the input for ONNX graph tracing.

+

The return value should be what expected by forward().

+
+ +
+ +
+
+class ml.rl.models.parametric_dqn.ParametricDQNWithPreprocessing(*args: Any, **kwargs: Any)
+

Bases: ml.rl.models.base.ModelBase

+
+
+forward(input)
+
+ +
+
+input_prototype()
+

This function provides the input for ONNX graph tracing.

+

The return value should be what expected by forward().

+
+ +
+ +
+
+

ml.rl.models.quantile_dqn module

+
+
+class ml.rl.models.quantile_dqn.QuantileDQN(*args: Any, **kwargs: Any)
+

Bases: ml.rl.models.base.ModelBase

+
+
+dist(input: ml.rl.types.PreprocessedState)
+
+ +
+
+forward(input: ml.rl.types.PreprocessedState)
+
+ +
+
+input_prototype()
+

This function provides the input for ONNX graph tracing.

+

The return value should be what expected by forward().

+
+ +
+ +
+
+

ml.rl.models.seq2slate module

+
+
+class ml.rl.models.seq2slate.BaselineNet(*args: Any, **kwargs: Any)
+

Bases: torch.nn.Module

+
+
+forward(input: ml.rl.types.PreprocessedRankingInput)
+
+ +
+ +
+
+class ml.rl.models.seq2slate.Decoder(*args: Any, **kwargs: Any)
+

Bases: torch.nn.Module

+

Generic num_layers layer decoder with masking.

+
+
+forward(x, memory, tgt_src_mask, tgt_tgt_mask)
+
+ +
+ +
+
+class ml.rl.models.seq2slate.DecoderLayer(*args: Any, **kwargs: Any)
+

Bases: torch.nn.Module

+

Decoder is made of self-attn, src-attn, and feed forward

+
+
+forward(x, m, tgt_src_mask, tgt_tgt_mask)
+
+ +
+ +
+
+class ml.rl.models.seq2slate.Embedder(*args: Any, **kwargs: Any)
+

Bases: torch.nn.Module

+
+
+forward(x)
+
+ +
+ +
+
+class ml.rl.models.seq2slate.Encoder(*args: Any, **kwargs: Any)
+

Bases: torch.nn.Module

+

Core encoder is a stack of num_layers layers

+
+
+forward(x, mask)
+

Pass the input (and mask) through each layer in turn.

+
+ +
+ +
+
+class ml.rl.models.seq2slate.EncoderLayer(*args: Any, **kwargs: Any)
+

Bases: torch.nn.Module

+

Encoder is made up of self-attn and feed forward

+
+
+forward(src_embed, src_mask)
+
+ +
+ +
+
+class ml.rl.models.seq2slate.Generator(*args: Any, **kwargs: Any)
+

Bases: torch.nn.Module

+

Define standard linear + softmax generation step.

+
+
+forward(mode, decoder_output=None, tgt_in_idx=None, greedy=None)
+
+ +
+ +
+
+class ml.rl.models.seq2slate.MultiHeadedAttention(*args: Any, **kwargs: Any)
+

Bases: torch.nn.Module

+
+
+forward(query, key, value, mask=None)
+
+ +
+ +
+
+class ml.rl.models.seq2slate.PositionalEncoding(*args: Any, **kwargs: Any)
+

Bases: torch.nn.Module

+
+
+forward(x, seq_len)
+
+ +
+ +
+
+class ml.rl.models.seq2slate.PositionwiseFeedForward(*args: Any, **kwargs: Any)
+

Bases: torch.nn.Module

+
+
+forward(x)
+
+ +
+ +
+
+class ml.rl.models.seq2slate.Seq2SlateMode(value)
+

Bases: enum.Enum

+

An enumeration.

+
+
+DECODE_ONE_STEP_MODE = 'decode_one_step'
+
+ +
+
+PER_SEQ_LOG_PROB_MODE = 'per_sequence_log_prob'
+
+ +
+
+PER_SYMBOL_LOG_PROB_DIST_MODE = 'per_symbol_log_prob_dist'
+
+ +
+
+RANK_MODE = 'rank'
+
+ +
+ +
+
+class ml.rl.models.seq2slate.Seq2SlateTransformerModel(*args: Any, **kwargs: Any)
+

Bases: torch.nn.Module

+

A Seq2Slate network with Transformer. The network is essentially an +encoder-decoder structure. The encoder inputs a sequence of candidate feature +vectors and a state feature vector, and the decoder outputs an ordered +list of candidate indices. The output order is learned through REINFORCE +algorithm to optimize some sequence-wise reward which is also specific to +the provided state feature.

+

One application example is to rank candidate feeds to a specific user such +that the final list of feeds as a whole optimizes the user’s engagement.

+

Seq2Slate paper: https://arxiv.org/abs/1810.02019 +Transformer paper: https://arxiv.org/abs/1706.03762

+
+
+decode(memory, state, tgt_src_mask, tgt_in_seq, tgt_tgt_mask, tgt_seq_len)
+
+ +
+
+encode(state, src_seq, src_mask)
+
+ +
+
+forward(input: ml.rl.types.PreprocessedRankingInput, mode: str, tgt_seq_len: Optional[int] = None, greedy: Optional[bool] = None)
+
+
Parameters
+
    +
  • input – model input

  • +
  • mode – a string indicating which mode to perform. +“rank”: return ranked actions and their generative probabilities. +“log_probs”: return generative log probabilities of given tgt sequences +(used for REINFORCE training)

  • +
  • tgt_seq_len – the length of output sequence to be decoded. Only used +in rank mode

  • +
  • greedy – whether to sample based on softmax distribution or greedily +when decoding. Only used in rank mode

  • +
+
+
+
+ +
+ +
+
+class ml.rl.models.seq2slate.Seq2SlateTransformerNet(*args: Any, **kwargs: Any)
+

Bases: ml.rl.models.base.ModelBase

+
+
+forward(input: ml.rl.types.PreprocessedRankingInput, mode: str, tgt_seq_len: Optional[int] = None, greedy: Optional[bool] = None)
+
+ +
+
+get_distributed_data_parallel_model()
+

Return DistributedDataParallel version of this model

+

This needs to be implemented explicitly because: +1) Model with EmbeddingBag module is not compatible with vanilla DistributedDataParallel +2) Exporting logic needs structured data. DistributedDataParallel doesn’t work with structured data.

+
+ +
+
+input_prototype()
+

This function provides the input for ONNX graph tracing.

+

The return value should be what expected by forward().

+
+ +
+ +
+
+class ml.rl.models.seq2slate.SublayerConnection(*args: Any, **kwargs: Any)
+

Bases: torch.nn.Module

+

A residual connection followed by a layer norm.

+
+
+forward(x, sublayer)
+
+ +
+ +
+
+ml.rl.models.seq2slate.attention(query, key, value, mask, d_k)
+

Scaled Dot Product Attention

+
+ +
+
+ml.rl.models.seq2slate.clones(module, N)
+

Produce N identical layers.

+
+
Parameters
+
    +
  • module – nn.Module class

  • +
  • N – number of copies

  • +
+
+
+
+ +
+
+ml.rl.models.seq2slate.subsequent_and_padding_mask(tgt_in_idx)
+

Create a mask to hide padding and future items

+
+ +
+
+ml.rl.models.seq2slate.subsequent_mask(size, device)
+

Mask out subsequent positions. Mainly used in the decoding process, +in which an item should not attend subsequent items.

+
+ +
+
+

ml.rl.models.seq2slate_reward module

+
+
+class ml.rl.models.seq2slate_reward.Seq2SlateRewardNet(*args: Any, **kwargs: Any)
+

Bases: ml.rl.models.base.ModelBase

+
+
+decode(memory, state, tgt_src_mask, tgt_in_seq, tgt_tgt_mask)
+

One step decoder. The decoder’s output will be used as the input to +the last layer for predicting slate reward

+
+ +
+
+encode(state, src_seq, src_mask)
+
+ +
+
+forward(input: ml.rl.types.PreprocessedRankingInput)
+

Encode tgt sequences and predict the slate reward.

+
+ +
+
+input_prototype()
+

This function provides the input for ONNX graph tracing.

+

The return value should be what expected by forward().

+
+ +
+ +
+
+class ml.rl.models.seq2slate_reward.Seq2SlateRewardNetJITWrapper(*args: Any, **kwargs: Any)
+

Bases: ml.rl.models.base.ModelBase

+
+
+forward(state: torch.Tensor, src_seq: torch.Tensor, tgt_out_seq: torch.Tensor, src_src_mask: torch.Tensor, slate_reward: torch.Tensor, tgt_out_idx: torch.Tensor) → torch.Tensor
+
+ +
+
+input_prototype(use_gpu=False)
+

This function provides the input for ONNX graph tracing.

+

The return value should be what expected by forward().

+
+ +
+ +
+
+

ml.rl.models.world_model module

+
+
+class ml.rl.models.world_model.MemoryNetwork(*args: Any, **kwargs: Any)
+

Bases: ml.rl.models.base.ModelBase

+
+
+forward(input)
+
+ +
+
+get_distributed_data_parallel_model()
+

Return DistributedDataParallel version of this model

+

This needs to be implemented explicitly because: +1) Model with EmbeddingBag module is not compatible with vanilla DistributedDataParallel +2) Exporting logic needs structured data. DistributedDataParallel doesn’t work with structured data.

+
+ +
+
+input_prototype()
+

This function provides the input for ONNX graph tracing.

+

The return value should be what expected by forward().

+
+ +
+ +
+
+

Module contents

+
+
+ + +
+
+ +
+
+
+
+ + + + \ No newline at end of file diff --git a/api/ml.rl.polyfill.html b/api/ml.rl.polyfill.html new file mode 100644 index 00000000..dd3495bd --- /dev/null +++ b/api/ml.rl.polyfill.html @@ -0,0 +1,188 @@ + + + + + + + ml.rl.polyfill package — ReAgent 1.0 documentation + + + + + + + + + + + + + + + + +
+ + +
+ +
+
+
+ +
+
+
+
+ +
+

ml.rl.polyfill package

+
+

Submodules

+
+
+

ml.rl.polyfill.decorators module

+
+
+class ml.rl.polyfill.decorators.ClassPropertyDescriptor(fget)
+

Bases: object

+
+ +
+
+ml.rl.polyfill.decorators.classproperty(func)
+

Allow for getter property usage on classmethods

+

cf: https://stackoverflow.com/questions/5189699/how-to-make-a-class-property

+
+ +
+
+

ml.rl.polyfill.exceptions module

+
+
+exception ml.rl.polyfill.exceptions.NoRetriesException
+

Bases: Exception

+
+ +
+
+exception ml.rl.polyfill.exceptions.NonRetryableException(child_exception, prefix='')
+

Bases: ml.rl.polyfill.exceptions.NoRetriesException

+
+ +
+
+exception ml.rl.polyfill.exceptions.NonRetryableTypeError(type_error)
+

Bases: TypeError, ml.rl.polyfill.exceptions.NonRetryableException

+
+ +
+
+

ml.rl.polyfill.types module

+

Polyfills fblearner types

+
+
+

Module contents

+
+
+ + +
+
+ +
+
+
+
+ + + + \ No newline at end of file diff --git a/api/ml.rl.prediction.html b/api/ml.rl.prediction.html new file mode 100644 index 00000000..44b60638 --- /dev/null +++ b/api/ml.rl.prediction.html @@ -0,0 +1,146 @@ + + + + + + + ml.rl.prediction package — ReAgent 1.0 documentation + + + + + + + + + + + + + + + + +
+ + +
+ +
+
+
+ +
+
+
+
+ +
+

ml.rl.prediction package

+
+

Submodules

+
+
+

ml.rl.prediction.dqn_torch_predictor module

+
+
+

ml.rl.prediction.predictor_wrapper module

+
+
+

Module contents

+
+
+ + +
+
+ +
+
+
+
+ + + + \ No newline at end of file diff --git a/api/ml.rl.preprocessing.html b/api/ml.rl.preprocessing.html new file mode 100644 index 00000000..9fe3fd17 --- /dev/null +++ b/api/ml.rl.preprocessing.html @@ -0,0 +1,167 @@ + + + + + + + ml.rl.preprocessing package — ReAgent 1.0 documentation + + + + + + + + + + + + + + + + +
+ + +
+ +
+
+
+ +
+
+
+
+ +
+

ml.rl.preprocessing package

+
+

Submodules

+
+
+

ml.rl.preprocessing.batch_preprocessor module

+
+
+

ml.rl.preprocessing.identify_types module

+
+
+ml.rl.preprocessing.identify_types.identify_type(values, enum_threshold=100)
+
+ +
+
+

ml.rl.preprocessing.normalization module

+
+
+

ml.rl.preprocessing.postprocessor module

+
+
+

ml.rl.preprocessing.preprocessor module

+
+
+

ml.rl.preprocessing.sparse_to_dense module

+
+
+

Module contents

+
+
+ + +
+
+ +
+
+
+
+ + + + \ No newline at end of file diff --git a/api/ml.rl.readers.html b/api/ml.rl.readers.html new file mode 100644 index 00000000..ba760b38 --- /dev/null +++ b/api/ml.rl.readers.html @@ -0,0 +1,312 @@ + + + + + + + ml.rl.readers package — ReAgent 1.0 documentation + + + + + + + + + + + + + + + + +
+ + +
+ +
+
+
+ +
+
+
+
+ +
+

ml.rl.readers package

+
+

Submodules

+
+
+

ml.rl.readers.base module

+
+
+class ml.rl.readers.base.ReaderBase(batch_size=None, drop_small=True, num_shards=None)
+

Bases: object

+
+
+do_get_shard(shard_id: int)
+

Subclass should implement this if the reader is shardable

+
+ +
+
+get_shard(shard_id: int)
+

Returns a shard of this reader

+
+ +
+ +
+
+class ml.rl.readers.base.ReaderIter
+

Bases: object

+
+
+abstract read_batch() → Optional[collections.OrderedDict]
+

Read a batch of data. The return value should be an OrderedDict. +Returns None when there is no more data.

+
+ +
+ +
+
+

ml.rl.readers.data_streamer module

+
+
+class ml.rl.readers.data_streamer.DataStreamer(data_reader, num_workers=0, pin_memory=False, timeout=0, worker_init_fn=None)
+

Bases: object

+

Data streamer. Provides single- or multi-process iterators over the data_reader.

+
+
Parameters
+
    +
  • data_reader (DataReader) – data_reader from which to stream the data.

  • +
  • num_workers (int, optional) – how many subprocesses to use for data +loading. 0 means that the data will be loaded in the main process. +(default: 0)

  • +
  • pin_memory (bool, optional) – If True, the data streamer will copy tensors +into CUDA pinned memory before returning them.

  • +
  • timeout (numeric, optional) – if positive, the timeout value for collecting a +batch from workers. Should always be non-negative. (default: 0)

  • +
  • worker_init_fn (callable, optional) – If not None, this will be called on each +worker subprocess with the worker id (an int in [0, num_workers - 1]) as +input, after seeding and before data loading. (default: None)

  • +
+
+
+
+

Note

+

By default, each worker will have its PyTorch seed set to +base_seed + worker_id, where base_seed is a long generated +by main process using its RNG. However, seeds for other libraries +may be duplicated upon initializing workers (w.g., NumPy), causing +each worker to return identical random numbers. (See +datastreamer-workers-random-seed section in FAQ.) You may +use torch.initial_seed() to access the PyTorch seed for each +worker in worker_init_fn, and use it to set other seeds +before data loading.

+
+
+

Warning

+

If spawn start method is used, worker_init_fn cannot be an +unpickleable object, e.g., a lambda function.

+
+
+ +
+
+class ml.rl.readers.data_streamer.WorkerDone(worker_id)
+

Bases: tuple

+
+
+worker_id
+

Alias for field number 0

+
+ +
+ +
+
+ml.rl.readers.data_streamer.pin_memory(batch)
+

This is ripped off from dataloader. The only difference is that it preserves +the type of Mapping so that the OrderedDict is maintained.

+
+ +
+
+

ml.rl.readers.json_dataset_reader module

+
+
+class ml.rl.readers.json_dataset_reader.JSONDatasetReader(path, batch_size=None, preprocess_handler=None)
+

Bases: ml.rl.readers.base.ReaderBase

+

Create the reader for a JSON training dataset.

+
+
+line_count()
+
+ +
+
+read_all()
+
+ +
+
+read_batch()
+
+ +
+
+reset_iterator()
+
+ +
+ +
+
+class ml.rl.readers.json_dataset_reader.JSONDatasetReaderIter(reader)
+

Bases: ml.rl.readers.base.ReaderIter

+
+
+read_batch() → Optional[collections.OrderedDict]
+

Read a batch of data. The return value should be an OrderedDict. +Returns None when there is no more data.

+
+ +
+ +
+
+

ml.rl.readers.nparray_reader module

+
+
+class ml.rl.readers.nparray_reader.NpArrayReader(data, size=None, **kwargs)
+

Bases: ml.rl.readers.base.ReaderBase

+

Basic reader taking np.ndarray`s of a whole dataset and split them into +chunks of `batch_size.

+
+
+do_get_shard(shard_id: int)
+

Subclass should implement this if the reader is shardable

+
+ +
+ +
+
+class ml.rl.readers.nparray_reader.NpArrayReaderIter(reader)
+

Bases: ml.rl.readers.base.ReaderIter

+
+
+read_batch() → Optional[collections.OrderedDict]
+

Read a batch of data. The return value should be an OrderedDict. +Returns None when there is no more data.

+
+ +
+ +
+
+

Module contents

+
+
+ + +
+
+ +
+
+
+
+ + + + \ No newline at end of file diff --git a/api/ml.rl.simulators.html b/api/ml.rl.simulators.html new file mode 100644 index 00000000..f794fd2e --- /dev/null +++ b/api/ml.rl.simulators.html @@ -0,0 +1,273 @@ + + + + + + + ml.rl.simulators package — ReAgent 1.0 documentation + + + + + + + + + + + + + + + + +
+ + +
+ +
+
+
+ +
+
+
+
+ +
+

ml.rl.simulators package

+
+

Submodules

+
+
+

ml.rl.simulators.recsim module

+
+
+class ml.rl.simulators.recsim.DocumentFeature(topic, length, quality)
+

Bases: NamedTuple

+
+
+as_vector()
+

Convenient function to get single tensor

+
+ +
+
+length: torch.Tensor
+

Alias for field number 1

+
+ +
+
+quality: torch.Tensor
+

Alias for field number 2

+
+ +
+
+topic: torch.Tensor
+

Alias for field number 0

+
+ +
+ +
+
+class ml.rl.simulators.recsim.RecSim(num_topics: int = 20, doc_length: float = 4, quality_means: List[Tuple[float, float]] = [(- 3.0, 0.0), (- 3.0, 0.0), (- 3.0, 0.0), (- 3.0, 0.0), (- 3.0, 0.0), (- 3.0, 0.0), (- 3.0, 0.0), (- 3.0, 0.0), (- 3.0, 0.0), (- 3.0, 0.0), (- 3.0, 0.0), (- 3.0, 0.0), (- 3.0, 0.0), (- 3.0, 0.0), (0.0, 3.0), (0.0, 3.0), (0.0, 3.0), (0.0, 3.0), (0.0, 3.0), (0.0, 3.0)], quality_variances: List[float] = [1.0, 1.0, 1.0, 1.0, 1.0, 1.0, 1.0, 1.0, 1.0, 1.0, 1.0, 1.0, 1.0, 1.0, 1.0, 1.0, 1.0, 1.0, 1.0, 1.0], initial_budget: float = 200, alpha: float = 1.0, m: int = 10, k: int = 3, num_users: int = 5000, y: float = 0.3, device: str = 'cpu', seed: int = 2147483647)
+

Bases: object

+

An environment described in Section 6 of https://arxiv.org/abs/1905.12767

+
+
+bonus(u, d, length, quality)
+
+ +
+
+compute_user_choice(slate: ml.rl.simulators.recsim.DocumentFeature) → Tuple[torch.Tensor, torch.Tensor]
+
+ +
+
+interest(u, d)
+
+
Parameters
+
    +
  • u – shape [batch, T]

  • +
  • d – shape [batch, k, T]

  • +
+
+
+
+ +
+
+obs() → Tuple[torch.Tensor, torch.Tensor, ml.rl.simulators.recsim.DocumentFeature]
+

Agent can observe: +- User interest vector +- Document topic vector +- Document length +- Document quality

+
+ +
+
+reset() → None
+
+ +
+
+rollout_policy(policy, memory_pool: Optional[ml.rl.test.gym.open_ai_gym_memory_pool.OpenAIGymMemoryPool] = None) → float
+
+ +
+
+sample_documents(n: int) → ml.rl.simulators.recsim.DocumentFeature
+
+ +
+
+sample_users(n)
+

User is represented by vector of topic interest, uniformly sampled from [-1, 1]

+
+ +
+
+satisfactory(u, d, quality)
+
+ +
+
+select(candidates: ml.rl.simulators.recsim.DocumentFeature, indices: torch.Tensor, add_null: bool) → ml.rl.simulators.recsim.DocumentFeature
+
+ +
+
+step(action: torch.Tensor) → Tuple[torch.Tensor, torch.Tensor, torch.Tensor, int]
+
+ +
+
+update_active_users() → int
+
+ +
+
+update_user_budget(selected_choice)
+
+ +
+
+update_user_interest(selected_choice)
+
+ +
+ +
+
+ml.rl.simulators.recsim.random_policy(obs: Tuple[torch.Tensor, torch.Tensor, ml.rl.simulators.recsim.DocumentFeature], recsim: ml.rl.simulators.recsim.RecSim)
+
+ +
+
+ml.rl.simulators.recsim.top_k_policy(q_network, obs: Tuple[torch.Tensor, torch.Tensor, ml.rl.simulators.recsim.DocumentFeature], recsim: ml.rl.simulators.recsim.RecSim)
+
+ +
+
+

Module contents

+
+
+ + +
+
+ +
+
+
+
+ + + + \ No newline at end of file diff --git a/api/ml.rl.training.gradient_free.html b/api/ml.rl.training.gradient_free.html new file mode 100644 index 00000000..a1887b70 --- /dev/null +++ b/api/ml.rl.training.gradient_free.html @@ -0,0 +1,174 @@ + + + + + + + ml.rl.training.gradient_free package — ReAgent 1.0 documentation + + + + + + + + + + + + + + + + +
+ + +
+ +
+
+
+ +
+
+
+
+ +
+

ml.rl.training.gradient_free package

+
+

Submodules

+
+
+

ml.rl.training.gradient_free.es_worker module

+
+
+

ml.rl.training.gradient_free.evolution_pool module

+
+
+

Module contents

+
+
+ + +
+
+ +
+
+
+
+ + + + \ No newline at end of file diff --git a/api/ml.rl.training.html b/api/ml.rl.training.html new file mode 100644 index 00000000..b1765db8 --- /dev/null +++ b/api/ml.rl.training.html @@ -0,0 +1,780 @@ + + + + + + + ml.rl.training package — ReAgent 1.0 documentation + + + + + + + + + + + + + + + + +
+ + +
+ +
+
+
+ +
+
+
+
+ +
+

ml.rl.training package

+
+

Subpackages

+ +
+
+

Submodules

+
+
+

ml.rl.training.c51_trainer module

+
+
+

ml.rl.training.cem_trainer module

+
+
+

ml.rl.training.dqn_trainer module

+
+
+

ml.rl.training.dqn_trainer_base module

+
+
+

ml.rl.training.imitator_training module

+
+
+

ml.rl.training.loss_reporter module

+
+
+class ml.rl.training.loss_reporter.BatchStats(td_loss, reward_loss, imitator_loss, logged_actions, logged_propensities, logged_rewards, logged_values, model_propensities, model_rewards, model_values, model_values_on_logged_actions, model_action_idxs)
+

Bases: NamedTuple

+
+
+static add_custom_scalars(action_names: Optional[List[str]])
+
+ +
+
+imitator_loss: Optional[torch.Tensor]
+

Alias for field number 2

+
+ +
+
+logged_actions: Optional[torch.Tensor]
+

Alias for field number 3

+
+ +
+
+logged_propensities: Optional[torch.Tensor]
+

Alias for field number 4

+
+ +
+
+logged_rewards: Optional[torch.Tensor]
+

Alias for field number 5

+
+ +
+
+logged_values: Optional[torch.Tensor]
+

Alias for field number 6

+
+ +
+
+model_action_idxs: Optional[torch.Tensor]
+

Alias for field number 11

+
+ +
+
+model_propensities: Optional[torch.Tensor]
+

Alias for field number 7

+
+ +
+
+model_rewards: Optional[torch.Tensor]
+

Alias for field number 8

+
+ +
+
+model_values: Optional[torch.Tensor]
+

Alias for field number 9

+
+ +
+
+model_values_on_logged_actions: Optional[torch.Tensor]
+

Alias for field number 10

+
+ +
+
+reward_loss: Optional[torch.Tensor]
+

Alias for field number 1

+
+ +
+
+td_loss: Optional[torch.Tensor]
+

Alias for field number 0

+
+ +
+
+write_summary(actions: List[str])
+
+ +
+ +
+
+class ml.rl.training.loss_reporter.LossReporter(action_names: Optional[List[str]] = None)
+

Bases: object

+
+
+RECENT_WINDOW_SIZE = 100
+
+ +
+
+static calculate_recent_window_average(arr, window_size, num_entries)
+
+ +
+
+flush()
+
+ +
+
+get_logged_action_distribution()
+
+ +
+
+get_model_action_distribution()
+
+ +
+
+get_recent_imitator_loss()
+
+ +
+
+get_recent_reward_loss()
+
+ +
+
+get_recent_rewards()
+
+ +
+
+get_recent_td_loss()
+
+ +
+
+get_td_loss_after_n(n)
+
+ +
+
+log_to_tensorboard(epoch: int) → None
+
+ +
+
+property num_batches
+
+ +
+
+report(**kwargs)
+
+ +
+ +
+
+class ml.rl.training.loss_reporter.StatsByAction(actions)
+

Bases: object

+
+
+append(stats)
+
+ +
+
+items()
+
+ +
+ +
+
+ml.rl.training.loss_reporter.merge_tensor_namedtuple_list(l, cls)
+
+ +
+
+

ml.rl.training.on_policy_predictor module

+
+
+class ml.rl.training.on_policy_predictor.CEMPlanningPredictor(trainer, action_dim: int, use_gpu: bool)
+

Bases: ml.rl.training.on_policy_predictor.OnPolicyPredictor

+
+
+discrete_action() → bool
+

Return True if this predictor is for a discrete action network

+
+ +
+
+policy(states: torch.Tensor, possible_actions_presence: Optional[torch.Tensor] = None) → Union[ml.rl.types.SacPolicyActionSet, ml.rl.types.DqnPolicyActionSet]
+
+ +
+
+policy_net() → bool
+

Return True if this predictor is for a policy network

+
+ +
+ +
+
+class ml.rl.training.on_policy_predictor.ContinuousActionOnPolicyPredictor(trainer, action_dim: int, use_gpu: bool)
+

Bases: ml.rl.training.on_policy_predictor.OnPolicyPredictor

+
+
+policy(states: torch.Tensor) → ml.rl.types.SacPolicyActionSet
+
+ +
+
+policy_net() → bool
+

Return True if this predictor is for a policy network

+
+ +
+ +
+
+class ml.rl.training.on_policy_predictor.DiscreteDQNOnPolicyPredictor(trainer, action_dim: int, use_gpu: bool)
+

Bases: ml.rl.training.on_policy_predictor.OnPolicyPredictor

+
+
+discrete_action() → bool
+

Return True if this predictor is for a discrete action network

+
+ +
+
+estimate_reward(state)
+
+ +
+
+policy(state: torch.Tensor, possible_actions_presence: torch.Tensor) → ml.rl.types.DqnPolicyActionSet
+
+ +
+
+policy_net() → bool
+

Return True if this predictor is for a policy network

+
+ +
+
+predict(state)
+
+ +
+ +
+
+class ml.rl.training.on_policy_predictor.OnPolicyPredictor(trainer, action_dim: int, use_gpu: bool)
+

Bases: object

+

This class generates actions given a trainer and a state. It’s used for +on-policy learning. If you have a TorchScript (i.e. serialized) model, +Use the classes in off_policy_predictor.py

+
+
+discrete_action() → bool
+

Return True if this predictor is for a discrete action network

+
+ +
+
+policy_net() → bool
+

Return True if this predictor is for a policy network

+
+ +
+ +
+
+class ml.rl.training.on_policy_predictor.ParametricDQNOnPolicyPredictor(trainer, action_dim: int, use_gpu: bool)
+

Bases: ml.rl.training.on_policy_predictor.OnPolicyPredictor

+
+
+discrete_action() → bool
+

Return True if this predictor is for a discrete action network

+
+ +
+
+estimate_reward(states_tiled: torch.Tensor, possible_actions: torch.Tensor)
+
+ +
+
+policy(states_tiled: torch.Tensor, possible_actions_with_presence: Tuple[torch.Tensor, torch.Tensor])
+
+ +
+
+policy_net() → bool
+

Return True if this predictor is for a policy network

+
+ +
+
+predict(states_tiled: torch.Tensor, possible_actions: torch.Tensor)
+
+ +
+ +
+
+

ml.rl.training.parametric_dqn_trainer module

+
+
+

ml.rl.training.qrdqn_trainer module

+
+
+

ml.rl.training.reward_network_trainer module

+
+
+class ml.rl.training.reward_network_trainer.RewardNetTrainer(reward_net: ml.rl.models.base.ModelBase, minibatch_size: int, use_gpu: bool = False, learning_rate: float = 0.001)
+

Bases: ml.rl.training.trainer.Trainer

+
+
+train(training_batch: ml.rl.types.PreprocessedTrainingBatch)
+
+ +
+
+warm_start_components()
+

The trainer should specify what members to save and load

+
+ +
+ +
+
+

ml.rl.training.rl_dataset module

+
+
+class ml.rl.training.rl_dataset.RLDataset(file_path)
+

Bases: object

+
+
+insert(**kwargs)
+
+ +
+
+insert_pre_timeline_format(mdp_id, sequence_number, state, timeline_format_action, reward, possible_actions, time_diff, action_probability, possible_actions_mask)
+

Insert a new sample to the dataset in the pre-timeline json format. +Format needed for running timeline operator and for uploading dataset to hive.

+
+ +
+
+insert_replay_buffer_format(state, action, reward, next_state, next_action, terminal, possible_next_actions, possible_next_actions_mask, time_diff, possible_actions, possible_actions_mask, policy_id)
+

Insert a new sample to the dataset in the same format as the +replay buffer.

+
+ +
+
+load()
+

Load samples from a gzipped json file.

+
+ +
+
+save()
+

Save samples as a pickle file or JSON file.

+
+ +
+ +
+
+

ml.rl.training.rl_trainer_pytorch module

+
+
+

ml.rl.training.sac_trainer module

+
+
+

ml.rl.training.slate_q_trainer module

+
+
+

ml.rl.training.td3_trainer module

+
+
+

ml.rl.training.trainer module

+
+
+class ml.rl.training.trainer.Trainer
+

Bases: object

+
+
+load_state_dict(state_dict)
+
+ +
+
+state_dict()
+
+ +
+
+train(training_batch: ml.rl.types.PreprocessedTrainingBatch) → None
+
+ +
+
+warm_start_components() → List[str]
+

The trainer should specify what members to save and load

+
+ +
+ +
+
+

ml.rl.training.training_data_page module

+
+
+class ml.rl.training.training_data_page.TrainingDataPage(mdp_ids: Optional[numpy.ndarray] = None, sequence_numbers: Optional[torch.Tensor] = None, states: Optional[torch.Tensor] = None, actions: Optional[torch.Tensor] = None, propensities: Optional[torch.Tensor] = None, rewards: Optional[torch.Tensor] = None, possible_actions_mask: Optional[torch.Tensor] = None, possible_actions_state_concat: Optional[torch.Tensor] = None, next_states: Optional[torch.Tensor] = None, next_actions: Optional[torch.Tensor] = None, possible_next_actions_mask: Optional[torch.Tensor] = None, possible_next_actions_state_concat: Optional[torch.Tensor] = None, not_terminal: Optional[torch.Tensor] = None, time_diffs: Optional[torch.Tensor] = None, metrics: Optional[torch.Tensor] = None, step: Optional[torch.Tensor] = None, max_num_actions: Optional[int] = None, next_propensities: Optional[torch.Tensor] = None, rewards_mask: Optional[torch.Tensor] = None)
+

Bases: object

+
+
+actions
+
+ +
+
+as_cem_training_batch(batch_first=False)
+

Generate one-step samples needed by CEM trainer. +The samples will be used to train an ensemble of world models used by CEM.

+
+
If batch_first = True:

state/next state shape: batch_size x 1 x state_dim +action shape: batch_size x 1 x action_dim +reward/terminal shape: batch_size x 1

+
+
else (default):

state/next state shape: 1 x batch_size x state_dim +action shape: 1 x batch_size x action_dim +reward/terminal shape: 1 x batch_size

+
+
+
+ +
+
+as_discrete_maxq_training_batch()
+
+ +
+
+as_parametric_maxq_training_batch()
+
+ +
+
+as_policy_network_training_batch()
+
+ +
+
+as_slate_q_training_batch()
+
+ +
+
+max_num_actions
+
+ +
+
+mdp_ids
+
+ +
+
+metrics
+
+ +
+
+next_actions
+
+ +
+
+next_propensities
+
+ +
+
+next_states
+
+ +
+
+not_terminal
+
+ +
+
+possible_actions_mask
+
+ +
+
+possible_actions_state_concat
+
+ +
+
+possible_next_actions_mask
+
+ +
+
+possible_next_actions_state_concat
+
+ +
+
+propensities
+
+ +
+
+rewards
+
+ +
+
+rewards_mask
+
+ +
+
+sequence_numbers
+
+ +
+
+set_device(device)
+
+ +
+
+set_type(dtype)
+
+ +
+
+size() → int
+
+ +
+
+states
+
+ +
+
+step
+
+ +
+
+time_diffs
+
+ +
+ +
+
+

Module contents

+
+
+ + +
+
+ +
+
+
+
+ + + + \ No newline at end of file diff --git a/api/ml.rl.training.ranking.html b/api/ml.rl.training.ranking.html new file mode 100644 index 00000000..4e9bd571 --- /dev/null +++ b/api/ml.rl.training.ranking.html @@ -0,0 +1,174 @@ + + + + + + + ml.rl.training.ranking package — ReAgent 1.0 documentation + + + + + + + + + + + + + + + + +
+ + +
+ +
+
+
+ +
+
+
+
+ +
+

ml.rl.training.ranking package

+
+

Submodules

+
+
+

ml.rl.training.ranking.seq2slate_tf_trainer module

+
+
+

ml.rl.training.ranking.seq2slate_trainer module

+
+
+

Module contents

+
+
+ + +
+
+ +
+
+
+
+ + + + \ No newline at end of file diff --git a/api/ml.rl.training.world_model.html b/api/ml.rl.training.world_model.html new file mode 100644 index 00000000..ad2fe139 --- /dev/null +++ b/api/ml.rl.training.world_model.html @@ -0,0 +1,170 @@ + + + + + + + ml.rl.training.world_model package — ReAgent 1.0 documentation + + + + + + + + + + + + + + + + +
+ + +
+ +
+
+
+ +
+
+
+
+ +
+

ml.rl.training.world_model package

+
+

Submodules

+
+
+

ml.rl.training.world_model.mdnrnn_trainer module

+
+
+

Module contents

+
+
+ + +
+
+ +
+
+
+
+ + + + \ No newline at end of file diff --git a/api/ml.rl.workflow.html b/api/ml.rl.workflow.html new file mode 100644 index 00000000..3f1f6937 --- /dev/null +++ b/api/ml.rl.workflow.html @@ -0,0 +1,170 @@ + + + + + + + ml.rl.workflow package — ReAgent 1.0 documentation + + + + + + + + + + + + + + + + +
+ + +
+ +
+
+
+ +
+
+
+
+ +
+

ml.rl.workflow package

+
+

Submodules

+
+
+

ml.rl.workflow.base_workflow module

+
+
+

ml.rl.workflow.create_normalization_metadata module

+
+
+

ml.rl.workflow.dqn_workflow module

+
+
+

ml.rl.workflow.helpers module

+
+
+

ml.rl.workflow.page_handler module

+
+
+

ml.rl.workflow.parametric_dqn_workflow module

+
+
+

ml.rl.workflow.preprocess_handler module

+
+
+

ml.rl.workflow.transitional module

+
+
+

Module contents

+
+
+ + +
+
+ +
+
+
+
+ + + + \ No newline at end of file diff --git a/api/reagent.training.cb.html b/api/reagent.training.cb.html new file mode 100644 index 00000000..32e0d9a0 --- /dev/null +++ b/api/reagent.training.cb.html @@ -0,0 +1,461 @@ + + + + + + + reagent.training.cb package — ReAgent 1.0 documentation + + + + + + + + + + + + + + + + +
+ + +
+ +
+
+
+ +
+
+
+
+ +
+

reagent.training.cb package

+
+

Submodules

+
+
+

reagent.training.cb.linucb_trainer module

+
+
+class reagent.training.cb.linucb_trainer.LinUCBTrainer(policy: reagent.gym.policies.policy.Policy, num_actions: int = - 1, use_interaction_features: bool = True)
+

Bases: reagent.training.reagent_lightning_module.ReAgentLightningModule

+

The trainer for LinUCB Contextual Bandit model. +The model estimates a ridge regression (linear) and only supports dense features. +The actions are assumed to be one of:

+
+
    +
  • +
    Fixed actions. The same (have the same semantic meaning) actions across all contexts.

    If actions are fixed, they can’t have features associated with them.

    +
    +
    +
  • +
  • +
    Feature actions. We can have different number and identities of actions in each

    context. The actions must have features to represent their semantic meaning.

    +
    +
    +
  • +
+
+

Reference: https://arxiv.org/pdf/1003.0146.pdf

+
+
Parameters
+
    +
  • policy – The policy to be trained. Its scorer has to be LinearRegressionUCB

  • +
  • num_actions – The number of actions. If num_actions==-1, the actions are assumed to be feature actions, +otherwise they are assumed to be fixed actions.

  • +
  • use_interaction_features – If True,

  • +
+
+
+
+
+allow_zero_length_dataloader_with_multiple_devices: bool
+
+ +
+
+configure_optimizers()
+

Choose what optimizers and learning-rate schedulers to use in your optimization. +Normally you’d need one. But in the case of GANs or similar you might have multiple.

+
+
Returns
+

Any of these 6 options.

+
    +
  • Single optimizer.

  • +
  • List or Tuple of optimizers.

  • +
  • Two lists - The first list has multiple optimizers, and the second has multiple LR schedulers +(or multiple lr_scheduler_config).

  • +
  • Dictionary, with an "optimizer" key, and (optionally) a "lr_scheduler" +key whose value is a single LR scheduler or lr_scheduler_config.

  • +
  • Tuple of dictionaries as described above, with an optional "frequency" key.

  • +
  • None - Fit will run without any optimizer.

  • +
+

+
+
+

The lr_scheduler_config is a dictionary which contains the scheduler and its associated configuration. +The default configuration is shown below.

+
lr_scheduler_config = {
+    # REQUIRED: The scheduler instance
+    "scheduler": lr_scheduler,
+    # The unit of the scheduler's step size, could also be 'step'.
+    # 'epoch' updates the scheduler on epoch end whereas 'step'
+    # updates it after a optimizer update.
+    "interval": "epoch",
+    # How many epochs/steps should pass between calls to
+    # `scheduler.step()`. 1 corresponds to updating the learning
+    # rate after every epoch/step.
+    "frequency": 1,
+    # Metric to to monitor for schedulers like `ReduceLROnPlateau`
+    "monitor": "val_loss",
+    # If set to `True`, will enforce that the value specified 'monitor'
+    # is available when the scheduler is updated, thus stopping
+    # training if not found. If set to `False`, it will only produce a warning
+    "strict": True,
+    # If using the `LearningRateMonitor` callback to monitor the
+    # learning rate progress, this keyword can be used to specify
+    # a custom logged name
+    "name": None,
+}
+
+
+

When there are schedulers in which the .step() method is conditioned on a value, such as the +torch.optim.lr_scheduler.ReduceLROnPlateau scheduler, Lightning requires that the +lr_scheduler_config contains the keyword "monitor" set to the metric name that the scheduler +should be conditioned on.

+

Metrics can be made available to monitor by simply logging it using +self.log('metric_to_track', metric_val) in your LightningModule.

+
+

Note

+

The frequency value specified in a dict along with the optimizer key is an int corresponding +to the number of sequential batches optimized with the specific optimizer. +It should be given to none or to all of the optimizers. +There is a difference between passing multiple optimizers in a list, +and passing multiple optimizers in dictionaries with a frequency of 1:

+
+
    +
  • In the former case, all optimizers will operate on the given batch in each optimization step.

  • +
  • In the latter, only one optimizer will operate on the given batch at every step.

  • +
+
+

This is different from the frequency value specified in the lr_scheduler_config mentioned above.

+
def configure_optimizers(self):
+    optimizer_one = torch.optim.SGD(self.model.parameters(), lr=0.01)
+    optimizer_two = torch.optim.SGD(self.model.parameters(), lr=0.01)
+    return [
+        {"optimizer": optimizer_one, "frequency": 5},
+        {"optimizer": optimizer_two, "frequency": 10},
+    ]
+
+
+

In this example, the first optimizer will be used for the first 5 steps, +the second optimizer for the next 10 steps and that cycle will continue. +If an LR scheduler is specified for an optimizer using the lr_scheduler key in the above dict, +the scheduler will only be updated when its optimizer is being used.

+
+

Examples:

+
# most cases. no learning rate scheduler
+def configure_optimizers(self):
+    return Adam(self.parameters(), lr=1e-3)
+
+# multiple optimizer case (e.g.: GAN)
+def configure_optimizers(self):
+    gen_opt = Adam(self.model_gen.parameters(), lr=0.01)
+    dis_opt = Adam(self.model_dis.parameters(), lr=0.02)
+    return gen_opt, dis_opt
+
+# example with learning rate schedulers
+def configure_optimizers(self):
+    gen_opt = Adam(self.model_gen.parameters(), lr=0.01)
+    dis_opt = Adam(self.model_dis.parameters(), lr=0.02)
+    dis_sch = CosineAnnealing(dis_opt, T_max=10)
+    return [gen_opt, dis_opt], [dis_sch]
+
+# example with step-based learning rate schedulers
+# each optimizer has its own scheduler
+def configure_optimizers(self):
+    gen_opt = Adam(self.model_gen.parameters(), lr=0.01)
+    dis_opt = Adam(self.model_dis.parameters(), lr=0.02)
+    gen_sch = {
+        'scheduler': ExponentialLR(gen_opt, 0.99),
+        'interval': 'step'  # called after each training step
+    }
+    dis_sch = CosineAnnealing(dis_opt, T_max=10) # called every epoch
+    return [gen_opt, dis_opt], [gen_sch, dis_sch]
+
+# example with optimizer frequencies
+# see training procedure in `Improved Training of Wasserstein GANs`, Algorithm 1
+# https://arxiv.org/abs/1704.00028
+def configure_optimizers(self):
+    gen_opt = Adam(self.model_gen.parameters(), lr=0.01)
+    dis_opt = Adam(self.model_dis.parameters(), lr=0.02)
+    n_critic = 5
+    return (
+        {'optimizer': dis_opt, 'frequency': n_critic},
+        {'optimizer': gen_opt, 'frequency': 1}
+    )
+
+
+
+

Note

+

Some things to know:

+
    +
  • Lightning calls .backward() and .step() on each optimizer and learning rate scheduler as needed.

  • +
  • If you use 16-bit precision (precision=16), Lightning will automatically handle the optimizers.

  • +
  • If you use multiple optimizers, training_step() will have an additional optimizer_idx parameter.

  • +
  • If you use torch.optim.LBFGS, Lightning handles the closure function automatically for you.

  • +
  • If you use multiple optimizers, gradients will be calculated only for the parameters of current optimizer +at each training step.

  • +
  • If you need to control how often those optimizers step or override the default .step() schedule, +override the optimizer_step() hook.

  • +
+
+
+ +
+
+precision: int
+
+ +
+
+prepare_data_per_node: bool
+
+ +
+
+training: bool
+
+ +
+
+training_step(batch: reagent.core.types.CBInput, batch_idx: int, optimizer_idx: int = 0)
+

Here you compute and return the training loss and some additional metrics for e.g. +the progress bar or logger.

+
+
Parameters
+
+
+
Returns
+

Any of.

+
    +
  • Tensor - The loss tensor

  • +
  • dict - A dictionary. Can include any keys, but must include the key 'loss'

  • +
  • +
    None - Training will skip to the next batch. This is only for automatic optimization.

    This is not supported for multi-GPU, TPU, IPU, or DeepSpeed.

    +
    +
    +
  • +
+

+
+
+

In this step you’d normally do the forward pass and calculate the loss for a batch. +You can also do fancier things like multiple forward passes or something model specific.

+

Example:

+
def training_step(self, batch, batch_idx):
+    x, y, z = batch
+    out = self.encoder(x)
+    loss = self.loss(out, x)
+    return loss
+
+
+

If you define multiple optimizers, this step will be called with an additional +optimizer_idx parameter.

+
# Multiple optimizers (e.g.: GANs)
+def training_step(self, batch, batch_idx, optimizer_idx):
+    if optimizer_idx == 0:
+        # do training_step with encoder
+        ...
+    if optimizer_idx == 1:
+        # do training_step with decoder
+        ...
+
+
+

If you add truncated back propagation through time you will also get an additional +argument with the hidden states of the previous step.

+
# Truncated back-propagation through time
+def training_step(self, batch, batch_idx, hiddens):
+    # hiddens are the hidden states from the previous truncated backprop step
+    out, hiddens = self.lstm(data, hiddens)
+    loss = ...
+    return {"loss": loss, "hiddens": hiddens}
+
+
+
+

Note

+

The loss value shown in the progress bar is smoothed (averaged) over the last values, +so it differs from the actual loss returned in train/validation step.

+
+
+ +
+
+update_params(x: torch.Tensor, y: torch.Tensor, weight: Optional[torch.Tensor] = None)
+
+
Parameters
+
    +
  • x – 2D tensor of shape (batch_size, dim)

  • +
  • y – 2D tensor of shape (batch_size, 1)

  • +
  • weight – 2D tensor of shape (batch_size, 1)

  • +
+
+
+
+ +
+
+use_amp: bool
+
+ +
+ +
+
+

Module contents

+
+
+ + +
+
+ +
+
+
+
+ + + + \ No newline at end of file