diff --git a/.DS_Store b/.DS_Store
new file mode 100644
index 00000000..45b2c008
Binary files /dev/null and b/.DS_Store differ
diff --git a/_sources/api/ml.rl.evaluation.rst.txt b/_sources/api/ml.rl.evaluation.rst.txt
new file mode 100644
index 00000000..b8b2c409
--- /dev/null
+++ b/_sources/api/ml.rl.evaluation.rst.txt
@@ -0,0 +1,85 @@
+ml.rl.evaluation package
+========================
+
+Submodules
+----------
+
+ml.rl.evaluation.cpe module
+---------------------------
+
+.. automodule:: ml.rl.evaluation.cpe
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.evaluation.doubly\_robust\_estimator module
+-------------------------------------------------
+
+.. automodule:: ml.rl.evaluation.doubly_robust_estimator
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.evaluation.evaluation\_data\_page module
+----------------------------------------------
+
+.. automodule:: ml.rl.evaluation.evaluation_data_page
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.evaluation.evaluator module
+---------------------------------
+
+.. automodule:: ml.rl.evaluation.evaluator
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.evaluation.ranking\_evaluator module
+------------------------------------------
+
+.. automodule:: ml.rl.evaluation.ranking_evaluator
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.evaluation.reward\_net\_evaluator module
+----------------------------------------------
+
+.. automodule:: ml.rl.evaluation.reward_net_evaluator
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.evaluation.sequential\_doubly\_robust\_estimator module
+-------------------------------------------------------------
+
+.. automodule:: ml.rl.evaluation.sequential_doubly_robust_estimator
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.evaluation.weighted\_sequential\_doubly\_robust\_estimator module
+-----------------------------------------------------------------------
+
+.. automodule:: ml.rl.evaluation.weighted_sequential_doubly_robust_estimator
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.evaluation.world\_model\_evaluator module
+-----------------------------------------------
+
+.. automodule:: ml.rl.evaluation.world_model_evaluator
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+Module contents
+---------------
+
+.. automodule:: ml.rl.evaluation
+ :members:
+ :undoc-members:
+ :show-inheritance:
diff --git a/_sources/api/ml.rl.models.rst.txt b/_sources/api/ml.rl.models.rst.txt
new file mode 100644
index 00000000..f4eda389
--- /dev/null
+++ b/_sources/api/ml.rl.models.rst.txt
@@ -0,0 +1,157 @@
+ml.rl.models package
+====================
+
+Submodules
+----------
+
+ml.rl.models.actor module
+-------------------------
+
+.. automodule:: ml.rl.models.actor
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.models.base module
+------------------------
+
+.. automodule:: ml.rl.models.base
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.models.bcq module
+-----------------------
+
+.. automodule:: ml.rl.models.bcq
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.models.categorical\_dqn module
+------------------------------------
+
+.. automodule:: ml.rl.models.categorical_dqn
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.models.cem\_planner module
+--------------------------------
+
+.. automodule:: ml.rl.models.cem_planner
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.models.convolutional\_network module
+------------------------------------------
+
+.. automodule:: ml.rl.models.convolutional_network
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.models.dqn module
+-----------------------
+
+.. automodule:: ml.rl.models.dqn
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.models.dueling\_q\_network module
+---------------------------------------
+
+.. automodule:: ml.rl.models.dueling_q_network
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.models.dueling\_quantile\_dqn module
+------------------------------------------
+
+.. automodule:: ml.rl.models.dueling_quantile_dqn
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.models.example\_sequence\_model module
+--------------------------------------------
+
+.. automodule:: ml.rl.models.example_sequence_model
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.models.fully\_connected\_network module
+---------------------------------------------
+
+.. automodule:: ml.rl.models.fully_connected_network
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.models.mdn\_rnn module
+----------------------------
+
+.. automodule:: ml.rl.models.mdn_rnn
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.models.no\_soft\_update\_embedding module
+-----------------------------------------------
+
+.. automodule:: ml.rl.models.no_soft_update_embedding
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.models.parametric\_dqn module
+-----------------------------------
+
+.. automodule:: ml.rl.models.parametric_dqn
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.models.quantile\_dqn module
+---------------------------------
+
+.. automodule:: ml.rl.models.quantile_dqn
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.models.seq2slate module
+-----------------------------
+
+.. automodule:: ml.rl.models.seq2slate
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.models.seq2slate\_reward module
+-------------------------------------
+
+.. automodule:: ml.rl.models.seq2slate_reward
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.models.world\_model module
+--------------------------------
+
+.. automodule:: ml.rl.models.world_model
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+Module contents
+---------------
+
+.. automodule:: ml.rl.models
+ :members:
+ :undoc-members:
+ :show-inheritance:
diff --git a/_sources/api/ml.rl.polyfill.rst.txt b/_sources/api/ml.rl.polyfill.rst.txt
new file mode 100644
index 00000000..0ef966a2
--- /dev/null
+++ b/_sources/api/ml.rl.polyfill.rst.txt
@@ -0,0 +1,37 @@
+ml.rl.polyfill package
+======================
+
+Submodules
+----------
+
+ml.rl.polyfill.decorators module
+--------------------------------
+
+.. automodule:: ml.rl.polyfill.decorators
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.polyfill.exceptions module
+--------------------------------
+
+.. automodule:: ml.rl.polyfill.exceptions
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.polyfill.types module
+---------------------------
+
+.. automodule:: ml.rl.polyfill.types
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+Module contents
+---------------
+
+.. automodule:: ml.rl.polyfill
+ :members:
+ :undoc-members:
+ :show-inheritance:
diff --git a/_sources/api/ml.rl.prediction.rst.txt b/_sources/api/ml.rl.prediction.rst.txt
new file mode 100644
index 00000000..551c92c9
--- /dev/null
+++ b/_sources/api/ml.rl.prediction.rst.txt
@@ -0,0 +1,29 @@
+ml.rl.prediction package
+========================
+
+Submodules
+----------
+
+ml.rl.prediction.dqn\_torch\_predictor module
+---------------------------------------------
+
+.. automodule:: ml.rl.prediction.dqn_torch_predictor
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.prediction.predictor\_wrapper module
+------------------------------------------
+
+.. automodule:: ml.rl.prediction.predictor_wrapper
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+Module contents
+---------------
+
+.. automodule:: ml.rl.prediction
+ :members:
+ :undoc-members:
+ :show-inheritance:
diff --git a/_sources/api/ml.rl.preprocessing.rst.txt b/_sources/api/ml.rl.preprocessing.rst.txt
new file mode 100644
index 00000000..61fe4a1b
--- /dev/null
+++ b/_sources/api/ml.rl.preprocessing.rst.txt
@@ -0,0 +1,61 @@
+ml.rl.preprocessing package
+===========================
+
+Submodules
+----------
+
+ml.rl.preprocessing.batch\_preprocessor module
+----------------------------------------------
+
+.. automodule:: ml.rl.preprocessing.batch_preprocessor
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.preprocessing.identify\_types module
+------------------------------------------
+
+.. automodule:: ml.rl.preprocessing.identify_types
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.preprocessing.normalization module
+----------------------------------------
+
+.. automodule:: ml.rl.preprocessing.normalization
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.preprocessing.postprocessor module
+----------------------------------------
+
+.. automodule:: ml.rl.preprocessing.postprocessor
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.preprocessing.preprocessor module
+---------------------------------------
+
+.. automodule:: ml.rl.preprocessing.preprocessor
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.preprocessing.sparse\_to\_dense module
+--------------------------------------------
+
+.. automodule:: ml.rl.preprocessing.sparse_to_dense
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+Module contents
+---------------
+
+.. automodule:: ml.rl.preprocessing
+ :members:
+ :undoc-members:
+ :show-inheritance:
diff --git a/_sources/api/ml.rl.readers.rst.txt b/_sources/api/ml.rl.readers.rst.txt
new file mode 100644
index 00000000..a81f6f7a
--- /dev/null
+++ b/_sources/api/ml.rl.readers.rst.txt
@@ -0,0 +1,45 @@
+ml.rl.readers package
+=====================
+
+Submodules
+----------
+
+ml.rl.readers.base module
+-------------------------
+
+.. automodule:: ml.rl.readers.base
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.readers.data\_streamer module
+-----------------------------------
+
+.. automodule:: ml.rl.readers.data_streamer
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.readers.json\_dataset\_reader module
+------------------------------------------
+
+.. automodule:: ml.rl.readers.json_dataset_reader
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.readers.nparray\_reader module
+------------------------------------
+
+.. automodule:: ml.rl.readers.nparray_reader
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+Module contents
+---------------
+
+.. automodule:: ml.rl.readers
+ :members:
+ :undoc-members:
+ :show-inheritance:
diff --git a/_sources/api/ml.rl.rst.txt b/_sources/api/ml.rl.rst.txt
new file mode 100644
index 00000000..6d8dca25
--- /dev/null
+++ b/_sources/api/ml.rl.rst.txt
@@ -0,0 +1,85 @@
+ml.rl package
+=============
+
+Subpackages
+-----------
+
+.. toctree::
+ :maxdepth: 4
+
+ ml.rl.evaluation
+ ml.rl.models
+ ml.rl.polyfill
+ ml.rl.prediction
+ ml.rl.preprocessing
+ ml.rl.readers
+ ml.rl.simulators
+ ml.rl.training
+ ml.rl.workflow
+
+Submodules
+----------
+
+ml.rl.caffe\_utils module
+-------------------------
+
+.. automodule:: ml.rl.caffe_utils
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.debug\_on\_error module
+-----------------------------
+
+.. automodule:: ml.rl.debug_on_error
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.json\_serialize module
+----------------------------
+
+.. automodule:: ml.rl.json_serialize
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.parameters module
+-----------------------
+
+.. automodule:: ml.rl.parameters
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.tensorboardX module
+-------------------------
+
+.. automodule:: ml.rl.tensorboardX
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.torch\_utils module
+-------------------------
+
+.. automodule:: ml.rl.torch_utils
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.types module
+------------------
+
+.. automodule:: ml.rl.types
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+Module contents
+---------------
+
+.. automodule:: ml.rl
+ :members:
+ :undoc-members:
+ :show-inheritance:
diff --git a/_sources/api/ml.rl.simulators.rst.txt b/_sources/api/ml.rl.simulators.rst.txt
new file mode 100644
index 00000000..fd0b0364
--- /dev/null
+++ b/_sources/api/ml.rl.simulators.rst.txt
@@ -0,0 +1,21 @@
+ml.rl.simulators package
+========================
+
+Submodules
+----------
+
+ml.rl.simulators.recsim module
+------------------------------
+
+.. automodule:: ml.rl.simulators.recsim
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+Module contents
+---------------
+
+.. automodule:: ml.rl.simulators
+ :members:
+ :undoc-members:
+ :show-inheritance:
diff --git a/_sources/api/ml.rl.training.gradient_free.rst.txt b/_sources/api/ml.rl.training.gradient_free.rst.txt
new file mode 100644
index 00000000..63d02ad8
--- /dev/null
+++ b/_sources/api/ml.rl.training.gradient_free.rst.txt
@@ -0,0 +1,29 @@
+ml.rl.training.gradient\_free package
+=====================================
+
+Submodules
+----------
+
+ml.rl.training.gradient\_free.es\_worker module
+-----------------------------------------------
+
+.. automodule:: ml.rl.training.gradient_free.es_worker
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.training.gradient\_free.evolution\_pool module
+----------------------------------------------------
+
+.. automodule:: ml.rl.training.gradient_free.evolution_pool
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+Module contents
+---------------
+
+.. automodule:: ml.rl.training.gradient_free
+ :members:
+ :undoc-members:
+ :show-inheritance:
diff --git a/_sources/api/ml.rl.training.ranking.rst.txt b/_sources/api/ml.rl.training.ranking.rst.txt
new file mode 100644
index 00000000..7cd69c56
--- /dev/null
+++ b/_sources/api/ml.rl.training.ranking.rst.txt
@@ -0,0 +1,29 @@
+ml.rl.training.ranking package
+==============================
+
+Submodules
+----------
+
+ml.rl.training.ranking.seq2slate\_tf\_trainer module
+----------------------------------------------------
+
+.. automodule:: ml.rl.training.ranking.seq2slate_tf_trainer
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.training.ranking.seq2slate\_trainer module
+------------------------------------------------
+
+.. automodule:: ml.rl.training.ranking.seq2slate_trainer
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+Module contents
+---------------
+
+.. automodule:: ml.rl.training.ranking
+ :members:
+ :undoc-members:
+ :show-inheritance:
diff --git a/_sources/api/ml.rl.training.rst.txt b/_sources/api/ml.rl.training.rst.txt
new file mode 100644
index 00000000..8b485924
--- /dev/null
+++ b/_sources/api/ml.rl.training.rst.txt
@@ -0,0 +1,159 @@
+ml.rl.training package
+======================
+
+Subpackages
+-----------
+
+.. toctree::
+ :maxdepth: 4
+
+ ml.rl.training.gradient_free
+ ml.rl.training.ranking
+ ml.rl.training.world_model
+
+Submodules
+----------
+
+ml.rl.training.c51\_trainer module
+----------------------------------
+
+.. automodule:: ml.rl.training.c51_trainer
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.training.cem\_trainer module
+----------------------------------
+
+.. automodule:: ml.rl.training.cem_trainer
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.training.dqn\_trainer module
+----------------------------------
+
+.. automodule:: ml.rl.training.dqn_trainer
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.training.dqn\_trainer\_base module
+----------------------------------------
+
+.. automodule:: ml.rl.training.dqn_trainer_base
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.training.imitator\_training module
+----------------------------------------
+
+.. automodule:: ml.rl.training.imitator_training
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.training.loss\_reporter module
+------------------------------------
+
+.. automodule:: ml.rl.training.loss_reporter
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.training.on\_policy\_predictor module
+-------------------------------------------
+
+.. automodule:: ml.rl.training.on_policy_predictor
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.training.parametric\_dqn\_trainer module
+----------------------------------------------
+
+.. automodule:: ml.rl.training.parametric_dqn_trainer
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.training.qrdqn\_trainer module
+------------------------------------
+
+.. automodule:: ml.rl.training.qrdqn_trainer
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.training.reward\_network\_trainer module
+----------------------------------------------
+
+.. automodule:: ml.rl.training.reward_network_trainer
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.training.rl\_dataset module
+---------------------------------
+
+.. automodule:: ml.rl.training.rl_dataset
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.training.rl\_trainer\_pytorch module
+------------------------------------------
+
+.. automodule:: ml.rl.training.rl_trainer_pytorch
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.training.sac\_trainer module
+----------------------------------
+
+.. automodule:: ml.rl.training.sac_trainer
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.training.slate\_q\_trainer module
+---------------------------------------
+
+.. automodule:: ml.rl.training.slate_q_trainer
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.training.td3\_trainer module
+----------------------------------
+
+.. automodule:: ml.rl.training.td3_trainer
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.training.trainer module
+-----------------------------
+
+.. automodule:: ml.rl.training.trainer
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.training.training\_data\_page module
+------------------------------------------
+
+.. automodule:: ml.rl.training.training_data_page
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+Module contents
+---------------
+
+.. automodule:: ml.rl.training
+ :members:
+ :undoc-members:
+ :show-inheritance:
diff --git a/_sources/api/ml.rl.training.world_model.rst.txt b/_sources/api/ml.rl.training.world_model.rst.txt
new file mode 100644
index 00000000..3797929c
--- /dev/null
+++ b/_sources/api/ml.rl.training.world_model.rst.txt
@@ -0,0 +1,21 @@
+ml.rl.training.world\_model package
+===================================
+
+Submodules
+----------
+
+ml.rl.training.world\_model.mdnrnn\_trainer module
+--------------------------------------------------
+
+.. automodule:: ml.rl.training.world_model.mdnrnn_trainer
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+Module contents
+---------------
+
+.. automodule:: ml.rl.training.world_model
+ :members:
+ :undoc-members:
+ :show-inheritance:
diff --git a/_sources/api/ml.rl.workflow.rst.txt b/_sources/api/ml.rl.workflow.rst.txt
new file mode 100644
index 00000000..2db46872
--- /dev/null
+++ b/_sources/api/ml.rl.workflow.rst.txt
@@ -0,0 +1,77 @@
+ml.rl.workflow package
+======================
+
+Submodules
+----------
+
+ml.rl.workflow.base\_workflow module
+------------------------------------
+
+.. automodule:: ml.rl.workflow.base_workflow
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.workflow.create\_normalization\_metadata module
+-----------------------------------------------------
+
+.. automodule:: ml.rl.workflow.create_normalization_metadata
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.workflow.dqn\_workflow module
+-----------------------------------
+
+.. automodule:: ml.rl.workflow.dqn_workflow
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.workflow.helpers module
+-----------------------------
+
+.. automodule:: ml.rl.workflow.helpers
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.workflow.page\_handler module
+-----------------------------------
+
+.. automodule:: ml.rl.workflow.page_handler
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.workflow.parametric\_dqn\_workflow module
+-----------------------------------------------
+
+.. automodule:: ml.rl.workflow.parametric_dqn_workflow
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.workflow.preprocess\_handler module
+-----------------------------------------
+
+.. automodule:: ml.rl.workflow.preprocess_handler
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+ml.rl.workflow.transitional module
+----------------------------------
+
+.. automodule:: ml.rl.workflow.transitional
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+Module contents
+---------------
+
+.. automodule:: ml.rl.workflow
+ :members:
+ :undoc-members:
+ :show-inheritance:
diff --git a/_sources/api/ml.rst.txt b/_sources/api/ml.rst.txt
new file mode 100644
index 00000000..fd94011c
--- /dev/null
+++ b/_sources/api/ml.rst.txt
@@ -0,0 +1,18 @@
+ml package
+==========
+
+Subpackages
+-----------
+
+.. toctree::
+ :maxdepth: 4
+
+ ml.rl
+
+Module contents
+---------------
+
+.. automodule:: ml
+ :members:
+ :undoc-members:
+ :show-inheritance:
diff --git a/_sources/api/reagent.training.cb.rst.txt b/_sources/api/reagent.training.cb.rst.txt
new file mode 100644
index 00000000..6484d74c
--- /dev/null
+++ b/_sources/api/reagent.training.cb.rst.txt
@@ -0,0 +1,21 @@
+reagent.training.cb package
+===========================
+
+Submodules
+----------
+
+reagent.training.cb.linucb\_trainer module
+------------------------------------------
+
+.. automodule:: reagent.training.cb.linucb_trainer
+ :members:
+ :undoc-members:
+ :show-inheritance:
+
+Module contents
+---------------
+
+.. automodule:: reagent.training.cb
+ :members:
+ :undoc-members:
+ :show-inheritance:
diff --git a/_static/.DS_Store b/_static/.DS_Store
new file mode 100644
index 00000000..f113a79e
Binary files /dev/null and b/_static/.DS_Store differ
diff --git a/api/ml.html b/api/ml.html
new file mode 100644
index 00000000..0abdf1a6
--- /dev/null
+++ b/api/ml.html
@@ -0,0 +1,278 @@
+
+
+
+
+
+
+ ml package — ReAgent 1.0 documentation
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
\ No newline at end of file
diff --git a/api/ml.rl.evaluation.html b/api/ml.rl.evaluation.html
new file mode 100644
index 00000000..2c8fb075
--- /dev/null
+++ b/api/ml.rl.evaluation.html
@@ -0,0 +1,314 @@
+
+
+
+
+
+
+ ml.rl.evaluation package — ReAgent 1.0 documentation
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+ ReAgent
+
+
+
+
+
+
+
+
+
+ml.rl.evaluation package
+
+
+ml.rl.evaluation.cpe module
+
+
+class ml.rl.evaluation.cpe. CpeDetails
+Bases: object
+
+
+log ( )
+
+
+
+
+log_to_tensorboard ( ) → None
+
+
+
+
+
+
+class ml.rl.evaluation.cpe. CpeEstimate ( raw , normalized , raw_std_error , normalized_std_error )
+Bases: NamedTuple
+
+
+normalized : float
+Alias for field number 1
+
+
+
+
+normalized_std_error : float
+Alias for field number 3
+
+
+
+
+raw : float
+Alias for field number 0
+
+
+
+
+raw_std_error : float
+Alias for field number 2
+
+
+
+
+
+
+class ml.rl.evaluation.cpe. CpeEstimateSet ( direct_method , inverse_propensity , doubly_robust , sequential_doubly_robust , weighted_doubly_robust , magic )
+Bases: NamedTuple
+
+
+check_estimates_exist ( )
+
+
+
+
+direct_method : Optional [ ml.rl.evaluation.cpe.CpeEstimate ]
+Alias for field number 0
+
+
+
+
+doubly_robust : Optional [ ml.rl.evaluation.cpe.CpeEstimate ]
+Alias for field number 2
+
+
+
+
+fill_empty_with_zero ( )
+
+
+
+
+inverse_propensity : Optional [ ml.rl.evaluation.cpe.CpeEstimate ]
+Alias for field number 1
+
+
+
+
+log ( )
+
+
+
+
+log_to_tensorboard ( metric_name : str ) → None
+
+
+
+
+magic : Optional [ ml.rl.evaluation.cpe.CpeEstimate ]
+Alias for field number 5
+
+
+
+
+sequential_doubly_robust : Optional [ ml.rl.evaluation.cpe.CpeEstimate ]
+Alias for field number 3
+
+
+
+
+weighted_doubly_robust : Optional [ ml.rl.evaluation.cpe.CpeEstimate ]
+Alias for field number 4
+
+
+
+
+
+
+ml.rl.evaluation.cpe. bootstrapped_std_error_of_mean ( data , sample_percent = 0.25 , num_samples = 1000 )
+Compute bootstrapped standard error of mean of input data.
+
+Parameters
+
+data – Input data (1D torch tensor or numpy array).
+sample_percent – Size of sample to use to calculate bootstrap statistic.
+num_samples – Number of times to sample.
+
+
+
+
+
+
+
+ml.rl.evaluation.doubly_robust_estimator module
+
+
+ml.rl.evaluation.evaluation_data_page module
+
+
+ml.rl.evaluation.evaluator module
+
+
+ml.rl.evaluation.ranking_evaluator module
+
+
+
+ml.rl.evaluation.sequential_doubly_robust_estimator module
+
+
+ml.rl.evaluation.weighted_sequential_doubly_robust_estimator module
+
+
+ml.rl.evaluation.world_model_evaluator module
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
\ No newline at end of file
diff --git a/api/ml.rl.html b/api/ml.rl.html
new file mode 100644
index 00000000..92a0790d
--- /dev/null
+++ b/api/ml.rl.html
@@ -0,0 +1,1404 @@
+
+
+
+
+
+
+ ml.rl package — ReAgent 1.0 documentation
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+ ReAgent
+
+
+
+
+
+
+
+
+
+ml.rl package
+
+
+
+ml.rl.caffe_utils module
+
+
+class ml.rl.caffe_utils. C2
+Bases: object
+
+
+static NextBlob ( prefix : str ) → str
+
+
+
+
+static init_net ( )
+
+
+
+
+static model ( )
+
+
+
+
+static net ( )
+
+
+
+
+static set_model ( model )
+
+
+
+
+static set_net ( net )
+
+
+
+
+static set_net_and_init_net ( net , init_net )
+
+
+
+
+
+
+class ml.rl.caffe_utils. C2Meta
+Bases: type
+
+
+
+
+ml.rl.debug_on_error module
+
+
+ml.rl.debug_on_error. start ( )
+
+
+
+
+ml.rl.json_serialize module
+
+
+ml.rl.json_serialize. from_json ( j_obj : Any , to_type : Type ) → Any
+
+
+
+
+ml.rl.json_serialize. isinstance_namedtuple ( x )
+
+
+
+
+ml.rl.json_serialize. json_to_object ( j : str , to_type : Type ) → Any
+
+
+
+
+ml.rl.json_serialize. object_to_json ( o : Any ) → str
+
+
+
+
+ml.rl.json_serialize. prepare_for_json ( o : Any ) → Any
+
+
+
+
+ml.rl.parameters module
+
+
+ml.rl.tensorboardX module
+Context library to allow dropping tensorboardX anywhere in the codebase.
+If there is no SummaryWriter in the context, function calls will be no-op.
+Usage:
+
+writer = SummaryWriter()
+
+with summary_writer_context(writer): some_func()
+
+def some_func(): SummaryWriterContext.add_scalar(“foo”, tensor)
+
+
+
+
+
+class ml.rl.tensorboardX. SummaryWriterContext
+Bases: object
+
+
+classmethod add_custom_scalars ( writer )
+Call this once you are satisfied setting up custom scalar
+
+
+
+
+classmethod add_custom_scalars_multilinechart ( tags , category = None , title = None )
+
+
+
+
+classmethod add_histogram ( key , val , * args , ** kwargs )
+
+
+
+
+classmethod increase_global_step ( )
+
+
+
+
+classmethod pop ( )
+
+
+
+
+classmethod push ( writer )
+
+
+
+
+
+
+class ml.rl.tensorboardX. SummaryWriterContextMeta
+Bases: type
+
+
+
+
+ml.rl.tensorboardX. summary_writer_context ( writer )
+
+
+
+
+ml.rl.torch_utils module
+
+
+ml.rl.torch_utils. export_module_to_buffer ( module ) → _io.BytesIO
+
+
+
+
+ml.rl.torch_utils. masked_softmax ( x , mask , temperature )
+Compute softmax values for each sets of scores in x.
+
+
+
+
+ml.rl.torch_utils. rescale_torch_tensor ( tensor : torch.Tensor , new_min : torch.Tensor , new_max : torch.Tensor , prev_min : torch.Tensor , prev_max : torch.Tensor )
+Rescale column values in N X M torch tensor to be in new range.
+Each column m in input tensor will be rescaled from range
+[prev_min[m], prev_max[m]] to [new_min[m], new_max[m]]
+
+
+
+
+ml.rl.torch_utils. softmax ( x , temperature )
+Compute softmax values for each sets of scores in x.
+
+
+
+
+ml.rl.torch_utils. stack ( mems )
+Stack a list of tensors
+Could use torch.stack here but torch.stack is much slower
+than torch.cat + view
+Submitted an issue for investigation:
+https://github.com/pytorch/pytorch/issues/22462
+FIXME: Remove this function after the issue above is resolved
+
+
+
+
+ml.rl.types module
+
+
+class ml.rl.types. ActorOutput ( action : torch.Tensor , log_prob : Optional [ torch.Tensor ] = None , action_mean : Optional [ torch.Tensor ] = None )
+Bases: ml.rl.types.BaseDataClass
+
+
+action : torch.Tensor
+
+
+
+
+action_mean : Optional [ torch.Tensor ] = None
+
+
+
+
+log_prob : Optional [ torch.Tensor ] = None
+
+
+
+
+
+
+class ml.rl.types. AllActionQValues ( q_values : torch.Tensor )
+Bases: ml.rl.types.BaseDataClass
+
+
+q_values : torch.Tensor
+
+
+
+
+
+
+class ml.rl.types. BaseDataClass
+Bases: object
+
+
+cuda ( )
+
+
+
+
+pin_memory ( )
+
+
+
+
+
+
+class ml.rl.types. CommonInput ( reward : torch.Tensor , time_diff : torch.Tensor , step : Optional [ torch.Tensor ] , not_terminal : torch.Tensor )
+Bases: ml.rl.types.BaseDataClass
+Base class for all inputs, both raw and preprocessed
+
+
+not_terminal : torch.Tensor
+
+
+
+
+reward : torch.Tensor
+
+
+
+
+step : Optional [ torch.Tensor ]
+
+
+
+
+time_diff : torch.Tensor
+
+
+
+
+
+
+class ml.rl.types. DqnPolicyActionSet ( greedy : int , softmax : Optional [ int ] = None , greedy_act_name : Optional [ str ] = None , softmax_act_name : Optional [ str ] = None , softmax_act_prob : Optional [ float ] = None )
+Bases: ml.rl.types.BaseDataClass
+
+
+greedy : int
+
+
+
+
+greedy_act_name : Optional [ str ] = None
+
+
+
+
+softmax : Optional [ int ] = None
+
+
+
+
+softmax_act_name : Optional [ str ] = None
+
+
+
+
+softmax_act_prob : Optional [ float ] = None
+
+
+
+
+
+
+Bases: ml.rl.types.BaseDataClass
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+class ml.rl.types. FeatureVector ( float_features: ml.rl.types.ValuePresence , id_list_features: Dict[str , Tuple[torch.Tensor , torch.Tensor]] = <factory> , time_since_first: Optional[torch.Tensor] = None )
+Bases: ml.rl.types.BaseDataClass
+
+
+float_features : ml.rl.types.ValuePresence
+
+
+
+
+id_list_features : Dict [ str , Tuple [ torch.Tensor , torch.Tensor ] ]
+
+
+
+
+time_since_first : Optional [ torch.Tensor ] = None
+
+
+
+
+
+
+class ml.rl.types. FloatFeatureInfo ( name : str , feature_id : int )
+Bases: ml.rl.types.BaseDataClass
+
+
+feature_id : int
+
+
+
+
+name : str
+
+
+
+
+
+
+class ml.rl.types. IdListFeatureConfig ( name : str , feature_id : int , id_mapping_name : str )
+Bases: ml.rl.types.BaseDataClass
+This describes how to map raw features to model features
+
+
+feature_id : int
+
+
+
+
+id_mapping_name : str
+
+
+
+
+name : str
+
+
+
+
+
+
+class ml.rl.types. IdMapping ( ids : List [ int ] )
+Bases: ml.rl.types.BaseDataClass
+
+
+ids : List [ int ]
+
+
+
+
+
+
+class ml.rl.types. MemoryNetworkOutput ( mus : torch.Tensor , sigmas : torch.Tensor , logpi : torch.Tensor , reward : torch.Tensor , not_terminal : torch.Tensor , last_step_lstm_hidden : torch.Tensor , last_step_lstm_cell : torch.Tensor , all_steps_lstm_hidden : torch.Tensor )
+Bases: ml.rl.types.BaseDataClass
+
+
+all_steps_lstm_hidden : torch.Tensor
+
+
+
+
+last_step_lstm_cell : torch.Tensor
+
+
+
+
+last_step_lstm_hidden : torch.Tensor
+
+
+
+
+logpi : torch.Tensor
+
+
+
+
+mus : torch.Tensor
+
+
+
+
+not_terminal : torch.Tensor
+
+
+
+
+reward : torch.Tensor
+
+
+
+
+sigmas : torch.Tensor
+
+
+
+
+
+
+class ml.rl.types. ModelFeatureConfig ( float_feature_infos: List[ml.rl.types.FloatFeatureInfo], id_mapping_config: Dict[str, ml.rl.types.IdMapping] = <factory>, id_list_feature_configs: List[ml.rl.types.IdListFeatureConfig] = <factory> )
+Bases: ml.rl.types.BaseDataClass
+
+
+float_feature_infos : List [ ml.rl.types.FloatFeatureInfo ]
+
+
+
+
+id_list_feature_configs : List [ ml.rl.types.IdListFeatureConfig ]
+
+
+
+
+id_mapping_config : Dict [ str , ml.rl.types.IdMapping ]
+
+
+
+
+
+
+class ml.rl.types. PlanningPolicyOutput ( next_best_continuous_action : Optional [ torch.Tensor ] = None , next_best_discrete_action_one_hot : Optional [ torch.Tensor ] = None , next_best_discrete_action_idx : Optional [ int ] = None )
+Bases: ml.rl.types.BaseDataClass
+
+
+next_best_continuous_action : Optional [ torch.Tensor ] = None
+
+
+
+
+next_best_discrete_action_idx : Optional [ int ] = None
+
+
+
+
+next_best_discrete_action_one_hot : Optional [ torch.Tensor ] = None
+
+
+
+
+
+
+class ml.rl.types. PreprocessedBaseInput ( reward : torch.Tensor , time_diff : torch.Tensor , step : Optional [ torch.Tensor ] , not_terminal : torch.Tensor , state : ml.rl.types.PreprocessedFeatureVector , next_state : ml.rl.types.PreprocessedFeatureVector )
+Bases: ml.rl.types.CommonInput
+
+
+next_state : ml.rl.types.PreprocessedFeatureVector
+
+
+
+
+state : ml.rl.types.PreprocessedFeatureVector
+
+
+
+
+
+
+class ml.rl.types. PreprocessedDiscreteDqnInput ( reward : torch.Tensor , time_diff : torch.Tensor , step : Optional [ torch.Tensor ] , not_terminal : torch.Tensor , state : ml.rl.types.PreprocessedFeatureVector , next_state : ml.rl.types.PreprocessedFeatureVector , action : torch.Tensor , next_action : torch.Tensor , possible_actions_mask : torch.Tensor , possible_next_actions_mask : torch.Tensor )
+Bases: ml.rl.types.PreprocessedBaseInput
+
+
+action : torch.Tensor
+
+
+
+
+next_action : torch.Tensor
+
+
+
+
+possible_actions_mask : torch.Tensor
+
+
+
+
+possible_next_actions_mask : torch.Tensor
+
+
+
+
+
+
+class ml.rl.types. PreprocessedFeatureVector ( float_features: torch.Tensor , id_list_features: Dict[str , Tuple[torch.Tensor , torch.Tensor]] = <factory> , time_since_first: Optional[torch.Tensor] = None )
+Bases: ml.rl.types.BaseDataClass
+
+
+float_features : torch.Tensor
+
+
+
+
+id_list_features : Dict [ str , Tuple [ torch.Tensor , torch.Tensor ] ]
+
+
+
+
+time_since_first : Optional [ torch.Tensor ] = None
+
+
+
+
+
+
+class ml.rl.types. PreprocessedMemoryNetworkInput ( reward : torch.Tensor , time_diff : torch.Tensor , step : Optional [ torch.Tensor ] , not_terminal : torch.Tensor , state : ml.rl.types.PreprocessedFeatureVector , next_state : ml.rl.types.PreprocessedFeatureVector , action : Union [ torch.Tensor , torch.Tensor ] )
+Bases: ml.rl.types.PreprocessedBaseInput
+
+
+action : Union [ torch.Tensor , torch.Tensor ]
+
+
+
+
+
+
+class ml.rl.types. PreprocessedParametricDqnInput ( reward : torch.Tensor , time_diff : torch.Tensor , step : Optional [ torch.Tensor ] , not_terminal : torch.Tensor , state : ml.rl.types.PreprocessedFeatureVector , next_state : ml.rl.types.PreprocessedFeatureVector , action : ml.rl.types.PreprocessedFeatureVector , next_action : ml.rl.types.PreprocessedFeatureVector , possible_actions : ml.rl.types.PreprocessedFeatureVector , possible_actions_mask : torch.Tensor , possible_next_actions : ml.rl.types.PreprocessedFeatureVector , possible_next_actions_mask : torch.Tensor , tiled_next_state : ml.rl.types.PreprocessedFeatureVector )
+Bases: ml.rl.types.PreprocessedBaseInput
+
+
+action : ml.rl.types.PreprocessedFeatureVector
+
+
+
+
+next_action : ml.rl.types.PreprocessedFeatureVector
+
+
+
+
+possible_actions : ml.rl.types.PreprocessedFeatureVector
+
+
+
+
+possible_actions_mask : torch.Tensor
+
+
+
+
+possible_next_actions : ml.rl.types.PreprocessedFeatureVector
+
+
+
+
+possible_next_actions_mask : torch.Tensor
+
+
+
+
+tiled_next_state : ml.rl.types.PreprocessedFeatureVector
+
+
+
+
+
+
+class ml.rl.types. PreprocessedPolicyNetworkInput ( reward : torch.Tensor , time_diff : torch.Tensor , step : Optional [ torch.Tensor ] , not_terminal : torch.Tensor , state : ml.rl.types.PreprocessedFeatureVector , next_state : ml.rl.types.PreprocessedFeatureVector , action : ml.rl.types.PreprocessedFeatureVector , next_action : ml.rl.types.PreprocessedFeatureVector )
+Bases: ml.rl.types.PreprocessedBaseInput
+
+
+action : ml.rl.types.PreprocessedFeatureVector
+
+
+
+
+next_action : ml.rl.types.PreprocessedFeatureVector
+
+
+
+
+
+
+class ml.rl.types. PreprocessedRankingInput ( state : ml.rl.types.PreprocessedFeatureVector , src_seq : ml.rl.types.PreprocessedFeatureVector , src_src_mask : torch.Tensor , tgt_in_seq : Optional [ ml.rl.types.PreprocessedFeatureVector ] = None , tgt_out_seq : Optional [ ml.rl.types.PreprocessedFeatureVector ] = None , tgt_tgt_mask : Optional [ torch.Tensor ] = None , slate_reward : Optional [ torch.Tensor ] = None , src_in_idx : Optional [ torch.Tensor ] = None , tgt_in_idx : Optional [ torch.Tensor ] = None , tgt_out_idx : Optional [ torch.Tensor ] = None , tgt_out_probs : Optional [ torch.Tensor ] = None , optim_tgt_in_idx : Optional [ torch.Tensor ] = None , optim_tgt_out_idx : Optional [ torch.Tensor ] = None , optim_tgt_in_seq : Optional [ ml.rl.types.PreprocessedFeatureVector ] = None , optim_tgt_out_seq : Optional [ ml.rl.types.PreprocessedFeatureVector ] = None )
+Bases: ml.rl.types.BaseDataClass
+
+
+classmethod from_tensors ( state : torch.Tensor , src_seq : torch.Tensor , src_src_mask : torch.Tensor , tgt_in_seq : Optional [ torch.Tensor ] = None , tgt_out_seq : Optional [ torch.Tensor ] = None , tgt_tgt_mask : Optional [ torch.Tensor ] = None , slate_reward : Optional [ torch.Tensor ] = None , src_in_idx : Optional [ torch.Tensor ] = None , tgt_in_idx : Optional [ torch.Tensor ] = None , tgt_out_idx : Optional [ torch.Tensor ] = None , tgt_out_probs : Optional [ torch.Tensor ] = None , optim_tgt_in_idx : Optional [ torch.Tensor ] = None , optim_tgt_out_idx : Optional [ torch.Tensor ] = None , optim_tgt_in_seq : Optional [ torch.Tensor ] = None , optim_tgt_out_seq : Optional [ torch.Tensor ] = None )
+
+
+
+
+optim_tgt_in_idx : Optional [ torch.Tensor ] = None
+
+
+
+
+optim_tgt_in_seq : Optional [ ml.rl.types.PreprocessedFeatureVector ] = None
+
+
+
+
+optim_tgt_out_idx : Optional [ torch.Tensor ] = None
+
+
+
+
+optim_tgt_out_seq : Optional [ ml.rl.types.PreprocessedFeatureVector ] = None
+
+
+
+
+slate_reward : Optional [ torch.Tensor ] = None
+
+
+
+
+src_in_idx : Optional [ torch.Tensor ] = None
+
+
+
+
+src_seq : ml.rl.types.PreprocessedFeatureVector
+
+
+
+
+src_src_mask : torch.Tensor
+
+
+
+
+state : ml.rl.types.PreprocessedFeatureVector
+
+
+
+
+tgt_in_idx : Optional [ torch.Tensor ] = None
+
+
+
+
+tgt_in_seq : Optional [ ml.rl.types.PreprocessedFeatureVector ] = None
+
+
+
+
+tgt_out_idx : Optional [ torch.Tensor ] = None
+
+
+
+
+tgt_out_probs : Optional [ torch.Tensor ] = None
+
+
+
+
+tgt_out_seq : Optional [ ml.rl.types.PreprocessedFeatureVector ] = None
+
+
+
+
+tgt_tgt_mask : Optional [ torch.Tensor ] = None
+
+
+
+
+
+
+class ml.rl.types. PreprocessedSlateFeatureVector ( float_features : torch.Tensor , item_mask : torch.Tensor , item_probability : torch.Tensor )
+Bases: ml.rl.types.BaseDataClass
+The shape of float_features is
+(batch_size, slate_size, item_dim) .
+item_mask masks available items in the action
+item_probability is the probability of item in being selected
+
+
+as_preprocessed_feature_vector ( ) → ml.rl.types.PreprocessedFeatureVector
+
+
+
+
+float_features : torch.Tensor
+
+
+
+
+item_mask : torch.Tensor
+
+
+
+
+item_probability : torch.Tensor
+
+
+
+
+
+
+class ml.rl.types. PreprocessedSlateQInput ( reward : torch.Tensor , time_diff : torch.Tensor , step : Optional [ torch.Tensor ] , not_terminal : torch.Tensor , state : ml.rl.types.PreprocessedFeatureVector , next_state : ml.rl.types.PreprocessedFeatureVector , tiled_state : ml.rl.types.PreprocessedTiledFeatureVector , tiled_next_state : ml.rl.types.PreprocessedTiledFeatureVector , action : ml.rl.types.PreprocessedSlateFeatureVector , next_action : ml.rl.types.PreprocessedSlateFeatureVector , reward_mask : torch.Tensor )
+Bases: ml.rl.types.PreprocessedBaseInput
+The shapes of tiled_state & tiled_next_state are
+(batch_size, slate_size, state_dim) .
+The shapes of reward , reward_mask , & next_item_mask are
+(batch_size, slate_size) .
+reward_mask indicated whether the reward could be observed, e.g.,
+the item got into viewport or not.
+
+
+action : ml.rl.types.PreprocessedSlateFeatureVector
+
+
+
+
+next_action : ml.rl.types.PreprocessedSlateFeatureVector
+
+
+
+
+reward_mask : torch.Tensor
+
+
+
+
+tiled_next_state : ml.rl.types.PreprocessedTiledFeatureVector
+
+
+
+
+tiled_state : ml.rl.types.PreprocessedTiledFeatureVector
+
+
+
+
+
+
+class ml.rl.types. PreprocessedState ( state )
+Bases: ml.rl.types.BaseDataClass
+This class makes it easier to plug modules into predictor
+
+
+classmethod from_tensor ( state : torch.Tensor )
+
+
+
+
+state : ml.rl.types.PreprocessedFeatureVector
+
+
+
+
+
+
+class ml.rl.types. PreprocessedStateAction ( state , action )
+Bases: ml.rl.types.BaseDataClass
+
+
+action : ml.rl.types.PreprocessedFeatureVector
+
+
+
+
+classmethod from_tensors ( state : torch.Tensor , action : torch.Tensor )
+
+
+
+
+state : ml.rl.types.PreprocessedFeatureVector
+
+
+
+
+
+
+class ml.rl.types. PreprocessedTiledFeatureVector ( float_features : torch.Tensor )
+Bases: ml.rl.types.BaseDataClass
+
+
+as_preprocessed_feature_vector ( ) → ml.rl.types.PreprocessedFeatureVector
+
+
+
+
+float_features : torch.Tensor
+
+
+
+
+
+
+class ml.rl.types. PreprocessedTrainingBatch ( training_input : Union [ ml.rl.types.PreprocessedBaseInput , ml.rl.types.PreprocessedDiscreteDqnInput , ml.rl.types.PreprocessedParametricDqnInput , ml.rl.types.PreprocessedMemoryNetworkInput , ml.rl.types.PreprocessedPolicyNetworkInput , ml.rl.types.PreprocessedRankingInput , ml.rl.types.PreprocessedSlateQInput ] , extras : Any )
+Bases: ml.rl.types.BaseDataClass
+
+
+batch_size ( )
+
+
+
+
+
+
+
+
+training_input : Union [ ml.rl.types.PreprocessedBaseInput , ml.rl.types.PreprocessedDiscreteDqnInput , ml.rl.types.PreprocessedParametricDqnInput , ml.rl.types.PreprocessedMemoryNetworkInput , ml.rl.types.PreprocessedPolicyNetworkInput , ml.rl.types.PreprocessedRankingInput , ml.rl.types.PreprocessedSlateQInput ]
+
+
+
+
+
+
+class ml.rl.types. RankingOutput ( ranked_tgt_out_idx : Optional [ torch.Tensor ] = None , ranked_tgt_out_probs : Optional [ torch.Tensor ] = None , log_probs : Optional [ torch.Tensor ] = None )
+Bases: ml.rl.types.BaseDataClass
+
+
+log_probs : Optional [ torch.Tensor ] = None
+
+
+
+
+ranked_tgt_out_idx : Optional [ torch.Tensor ] = None
+
+
+
+
+ranked_tgt_out_probs : Optional [ torch.Tensor ] = None
+
+
+
+
+
+
+class ml.rl.types. RawBaseInput ( reward : torch.Tensor , time_diff : torch.Tensor , step : Optional [ torch.Tensor ] , not_terminal : torch.Tensor , state : ml.rl.types.FeatureVector , next_state : ml.rl.types.FeatureVector )
+Bases: ml.rl.types.CommonInput
+
+
+next_state : ml.rl.types.FeatureVector
+
+
+
+
+state : ml.rl.types.FeatureVector
+
+
+
+
+
+
+class ml.rl.types. RawDiscreteDqnInput ( reward : torch.Tensor , time_diff : torch.Tensor , step : Optional [ torch.Tensor ] , not_terminal : torch.Tensor , state : ml.rl.types.FeatureVector , next_state : ml.rl.types.FeatureVector , action : torch.Tensor , next_action : torch.Tensor , possible_actions_mask : torch.Tensor , possible_next_actions_mask : torch.Tensor )
+Bases: ml.rl.types.RawBaseInput
+
+
+action : torch.Tensor
+
+
+
+
+next_action : torch.Tensor
+
+
+
+
+possible_actions_mask : torch.Tensor
+
+
+
+
+possible_next_actions_mask : torch.Tensor
+
+
+
+
+preprocess ( state : ml.rl.types.PreprocessedFeatureVector , next_state : ml.rl.types.PreprocessedFeatureVector )
+
+
+
+
+preprocess_tensors ( state : torch.Tensor , next_state : torch.Tensor )
+
+
+
+
+
+
+class ml.rl.types. RawMemoryNetworkInput ( reward : torch.Tensor , time_diff : torch.Tensor , step : Optional [ torch.Tensor ] , not_terminal : torch.Tensor , state : ml.rl.types.FeatureVector , next_state : ml.rl.types.FeatureVector , action : Union [ ml.rl.types.FeatureVector , torch.Tensor ] )
+Bases: ml.rl.types.RawBaseInput
+
+
+action : Union [ ml.rl.types.FeatureVector , torch.Tensor ]
+
+
+
+
+preprocess ( state : ml.rl.types.PreprocessedFeatureVector , next_state : ml.rl.types.PreprocessedFeatureVector , action : Optional [ torch.Tensor ] = None )
+
+
+
+
+preprocess_tensors ( state : torch.Tensor , next_state : torch.Tensor , action : Optional [ torch.Tensor ] = None )
+
+
+
+
+
+
+class ml.rl.types. RawParametricDqnInput ( reward : torch.Tensor , time_diff : torch.Tensor , step : Optional [ torch.Tensor ] , not_terminal : torch.Tensor , state : ml.rl.types.FeatureVector , next_state : ml.rl.types.FeatureVector , action : ml.rl.types.FeatureVector , next_action : ml.rl.types.FeatureVector , possible_actions : ml.rl.types.FeatureVector , possible_actions_mask : torch.Tensor , possible_next_actions : ml.rl.types.FeatureVector , possible_next_actions_mask : torch.Tensor , tiled_next_state : ml.rl.types.FeatureVector )
+Bases: ml.rl.types.RawBaseInput
+
+
+action : ml.rl.types.FeatureVector
+
+
+
+
+next_action : ml.rl.types.FeatureVector
+
+
+
+
+possible_actions : ml.rl.types.FeatureVector
+
+
+
+
+possible_actions_mask : torch.Tensor
+
+
+
+
+possible_next_actions : ml.rl.types.FeatureVector
+
+
+
+
+possible_next_actions_mask : torch.Tensor
+
+
+
+
+preprocess ( state : ml.rl.types.PreprocessedFeatureVector , next_state : ml.rl.types.PreprocessedFeatureVector , action : ml.rl.types.PreprocessedFeatureVector , next_action : ml.rl.types.PreprocessedFeatureVector , possible_actions : ml.rl.types.PreprocessedFeatureVector , possible_next_actions : ml.rl.types.PreprocessedFeatureVector , tiled_next_state : ml.rl.types.PreprocessedFeatureVector )
+
+
+
+
+preprocess_tensors ( state : torch.Tensor , next_state : torch.Tensor , action : torch.Tensor , next_action : torch.Tensor , possible_actions : torch.Tensor , possible_next_actions : torch.Tensor , tiled_next_state : torch.Tensor )
+
+
+
+
+tiled_next_state : ml.rl.types.FeatureVector
+
+
+
+
+
+
+class ml.rl.types. RawPolicyNetworkInput ( reward : torch.Tensor , time_diff : torch.Tensor , step : Optional [ torch.Tensor ] , not_terminal : torch.Tensor , state : ml.rl.types.FeatureVector , next_state : ml.rl.types.FeatureVector , action : ml.rl.types.FeatureVector , next_action : ml.rl.types.FeatureVector )
+Bases: ml.rl.types.RawBaseInput
+
+
+action : ml.rl.types.FeatureVector
+
+
+
+
+next_action : ml.rl.types.FeatureVector
+
+
+
+
+preprocess ( state : ml.rl.types.PreprocessedFeatureVector , next_state : ml.rl.types.PreprocessedFeatureVector , action : ml.rl.types.PreprocessedFeatureVector , next_action : ml.rl.types.PreprocessedFeatureVector )
+
+
+
+
+preprocess_tensors ( state : torch.Tensor , next_state : torch.Tensor , action : torch.Tensor , next_action : torch.Tensor )
+
+
+
+
+
+
+class ml.rl.types. RawStateAction ( state : ml.rl.types.FeatureVector , action : ml.rl.types.FeatureVector )
+Bases: ml.rl.types.BaseDataClass
+
+
+action : ml.rl.types.FeatureVector
+
+
+
+
+state : ml.rl.types.FeatureVector
+
+
+
+
+
+
+class ml.rl.types. RawTrainingBatch ( training_input : Union [ ml.rl.types.RawBaseInput , ml.rl.types.RawDiscreteDqnInput , ml.rl.types.RawParametricDqnInput , ml.rl.types.RawPolicyNetworkInput ] , extras : Any )
+Bases: ml.rl.types.BaseDataClass
+
+
+batch_size ( )
+
+
+
+
+
+
+
+
+preprocess ( training_input : Union [ ml.rl.types.PreprocessedBaseInput , ml.rl.types.PreprocessedDiscreteDqnInput , ml.rl.types.PreprocessedParametricDqnInput , ml.rl.types.PreprocessedMemoryNetworkInput , ml.rl.types.PreprocessedPolicyNetworkInput ] ) → ml.rl.types.PreprocessedTrainingBatch
+
+
+
+
+training_input : Union [ ml.rl.types.RawBaseInput , ml.rl.types.RawDiscreteDqnInput , ml.rl.types.RawParametricDqnInput , ml.rl.types.RawPolicyNetworkInput ]
+
+
+
+
+
+
+class ml.rl.types. RewardNetworkOutput ( predicted_reward : torch.Tensor )
+Bases: ml.rl.types.BaseDataClass
+
+
+predicted_reward : torch.Tensor
+
+
+
+
+
+
+class ml.rl.types. SacPolicyActionSet ( greedy : torch.Tensor , greedy_propensity : float )
+Bases: ml.rl.types.BaseDataClass
+
+
+greedy : torch.Tensor
+
+
+
+
+greedy_propensity : float
+
+
+
+
+
+
+class ml.rl.types. SingleQValue ( q_value : torch.Tensor )
+Bases: ml.rl.types.BaseDataClass
+
+
+q_value : torch.Tensor
+
+
+
+
+
+
+class ml.rl.types. ValuePresence ( value : torch.Tensor , presence : Optional [ torch.Tensor ] )
+Bases: ml.rl.types.BaseDataClass
+
+
+presence : Optional [ torch.Tensor ]
+
+
+
+
+value : torch.Tensor
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
\ No newline at end of file
diff --git a/api/ml.rl.models.html b/api/ml.rl.models.html
new file mode 100644
index 00000000..3e879ea7
--- /dev/null
+++ b/api/ml.rl.models.html
@@ -0,0 +1,1246 @@
+
+
+
+
+
+
+ ml.rl.models package — ReAgent 1.0 documentation
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+ ReAgent
+
+
+
+
+
+
+
+
+
+ml.rl.models package
+
+
+ml.rl.models.actor module
+
+
+class ml.rl.models.actor. DirichletFullyConnectedActor ( * args : Any , ** kwargs : Any )
+Bases: ml.rl.models.base.ModelBase
+
+
+EPSILON = 1e-06
+
+
+
+
+forward ( input )
+
+
+
+
+get_log_prob ( state , action )
+
+
+
+
+input_prototype ( )
+This function provides the input for ONNX graph tracing.
+The return value should be what expected by forward() .
+
+
+
+
+
+
+class ml.rl.models.actor. FullyConnectedActor ( * args : Any , ** kwargs : Any )
+Bases: ml.rl.models.base.ModelBase
+
+
+forward ( input )
+
+
+
+
+input_prototype ( )
+This function provides the input for ONNX graph tracing.
+The return value should be what expected by forward() .
+
+
+
+
+
+
+class ml.rl.models.actor. GaussianFullyConnectedActor ( * args : Any , ** kwargs : Any )
+Bases: ml.rl.models.base.ModelBase
+
+
+forward ( input )
+
+
+
+
+get_log_prob ( state , squashed_action )
+Action is expected to be squashed with tanh
+
+
+
+
+input_prototype ( )
+This function provides the input for ONNX graph tracing.
+The return value should be what expected by forward() .
+
+
+
+
+
+
+ml.rl.models.base module
+
+
+class ml.rl.models.base. ModelBase ( * args : Any , ** kwargs : Any )
+Bases: torch.nn.Module
+A base class to support exporting through ONNX
+
+
+cpu_model ( )
+Override this in DistributedDataParallel models
+
+
+
+
+feature_config ( ) → Optional [ ml.rl.types.ModelFeatureConfig ]
+If the model needs additional preprocessing, e.g., using sequence features,
+returns the config here.
+
+
+
+
+get_distributed_data_parallel_model ( )
+Return DistributedDataParallel version of this model
+This needs to be implemented explicitly because:
+1) Model with EmbeddingBag module is not compatible with vanilla DistributedDataParallel
+2) Exporting logic needs structured data. DistributedDataParallel doesn’t work with structured data.
+
+
+
+
+get_target_network ( )
+Return a copy of this network to be used as target network
+Subclass should override this if the target network should share parameters
+with the network to be trained.
+
+
+
+
+input_prototype ( ) → Any
+This function provides the input for ONNX graph tracing.
+The return value should be what expected by forward() .
+
+
+
+
+
+
+ml.rl.models.bcq module
+
+
+class ml.rl.models.bcq. BatchConstrainedDQN ( * args : Any , ** kwargs : Any )
+Bases: ml.rl.models.base.ModelBase
+
+
+forward ( input )
+
+
+
+
+input_prototype ( )
+This function provides the input for ONNX graph tracing.
+The return value should be what expected by forward() .
+
+
+
+
+
+
+ml.rl.models.categorical_dqn module
+
+
+class ml.rl.models.categorical_dqn. CategoricalDQN ( * args : Any , ** kwargs : Any )
+Bases: ml.rl.models.base.ModelBase
+
+
+forward ( input : ml.rl.types.PreprocessedState )
+
+
+
+
+input_prototype ( )
+This function provides the input for ONNX graph tracing.
+The return value should be what expected by forward() .
+
+
+
+
+log_dist ( input : ml.rl.types.PreprocessedState )
+
+
+
+
+serving_model ( )
+
+
+
+
+
+
+ml.rl.models.cem_planner module
+A network which implements a cross entropy method-based planner
+The planner plans the best next action based on simulation data generated by
+an ensemble of world models.
+The idea is inspired by: https://arxiv.org/abs/1805.12114
+
+
+class ml.rl.models.cem_planner. CEMPlanner ( * args : Any , ** kwargs : Any )
+Bases: ml.rl.models.base.ModelBase
+
+
+forward ( input : ml.rl.types.PreprocessedState )
+
+
+
+
+get_distributed_data_parallel_model ( )
+Return DistributedDataParallel version of this model
+This needs to be implemented explicitly because:
+1) Model with EmbeddingBag module is not compatible with vanilla DistributedDataParallel
+2) Exporting logic needs structured data. DistributedDataParallel doesn’t work with structured data.
+
+
+
+
+input_prototype ( )
+This function provides the input for ONNX graph tracing.
+The return value should be what expected by forward() .
+
+
+
+
+
+
+class ml.rl.models.cem_planner. CEMPlannerNetwork ( mem_net_list : List [ ml.rl.models.world_model.MemoryNetwork ] , cem_num_iterations : int , cem_population_size : int , ensemble_population_size : int , num_elites : int , plan_horizon_length : int , state_dim : int , action_dim : int , discrete_action : bool , terminal_effective : bool , gamma : float , alpha : float = 0.25 , epsilon : float = 0.001 , action_upper_bounds : Optional [ numpy.ndarray ] = None , action_lower_bounds : Optional [ numpy.ndarray ] = None )
+Bases: torch.nn.modules.module.Module
+
+
+acc_rewards_of_all_solutions ( input : ml.rl.types.PreprocessedState , solutions : torch.Tensor ) → float
+Calculate accumulated rewards of solutions.
+
+Parameters
+
+input – the input which contains the starting state
+solutions – its shape is (cem_pop_size, plan_horizon_length, action_dim)
+
+
+Returns
+a vector of size cem_pop_size, which is the reward of each solution
+
+
+
+
+
+
+acc_rewards_of_one_solution ( init_state : torch.Tensor , solution : torch.Tensor , solution_idx : int )
+ensemble_pop_size trajectories will be sampled to evaluate a
+CEM solution. Each trajectory is generated by one world model
+
+Parameters
+
+init_state – its shape is (state_dim, )
+solution – its shape is (plan_horizon_length, action_dim)
+solution_idx – the index of the solution
+
+
+Return reward
+Reward of each of ensemble_pop_size trajectories
+
+
+
+
+
+
+constrained_variance ( mean , var )
+
+
+
+
+continuous_planning ( input : ml.rl.types.PreprocessedState ) → numpy.ndarray
+
+
+
+
+discrete_planning ( input : ml.rl.types.PreprocessedState ) → Tuple [ int , numpy.ndarray ]
+
+
+
+
+forward ( input : ml.rl.types.PreprocessedState )
+Defines the computation performed at every call.
+Should be overridden by all subclasses.
+
+
Note
+
Although the recipe for forward pass needs to be defined within
+this function, one should call the Module instance afterwards
+instead of this since the former takes care of running the
+registered hooks while the latter silently ignores them.
+
+
+
+
+
+sample_reward_next_state_terminal ( world_model_input : ml.rl.types.PreprocessedStateAction , mem_net : ml.rl.models.world_model.MemoryNetwork )
+Sample one-step dynamics based on the provided world model
+
+
+
+
+training : bool
+
+
+
+
+
+
+ml.rl.models.convolutional_network module
+
+
+class ml.rl.models.convolutional_network. ConvolutionalNetwork ( cnn_parameters , layers , activations )
+Bases: torch.nn.modules.module.Module
+
+
+conv_forward ( input )
+
+
+
+
+forward ( input ) → torch.FloatTensor
+Forward pass for generic convnet DNNs. Assumes activation names
+are valid pytorch activation names.
+:param input image tensor
+
+
+
+
+training : bool
+
+
+
+
+
+
+ml.rl.models.dqn module
+
+
+class ml.rl.models.dqn. FullyConnectedDQN ( * args : Any , ** kwargs : Any )
+Bases: ml.rl.models.base.ModelBase
+
+
+forward ( input : ml.rl.types.PreprocessedState )
+
+
+
+
+get_distributed_data_parallel_model ( )
+Return DistributedDataParallel version of this model
+This needs to be implemented explicitly because:
+1) Model with EmbeddingBag module is not compatible with vanilla DistributedDataParallel
+2) Exporting logic needs structured data. DistributedDataParallel doesn’t work with structured data.
+
+
+
+
+input_prototype ( )
+This function provides the input for ONNX graph tracing.
+The return value should be what expected by forward() .
+
+
+
+
+
+
+ml.rl.models.dueling_q_network module
+
+
+class ml.rl.models.dueling_q_network. DuelingQNetwork ( * args : Any , ** kwargs : Any )
+Bases: ml.rl.models.base.ModelBase
+
+
+forward ( input ) → Union [ NamedTuple , torch.FloatTensor ]
+
+
+
+
+input_prototype ( )
+This function provides the input for ONNX graph tracing.
+The return value should be what expected by forward() .
+
+
+
+
+
+
+
+ml.rl.models.example_sequence_model module
+
+
+class ml.rl.models.example_sequence_model. ExampleSequenceModel ( * args : Any , ** kwargs : Any )
+Bases: ml.rl.models.base.ModelBase
+
+
+feature_config ( )
+If the model needs additional preprocessing, e.g., using sequence features,
+returns the config here.
+
+
+
+
+forward ( state )
+
+
+
+
+input_prototype ( )
+This function provides the input for ONNX graph tracing.
+The return value should be what expected by forward() .
+
+
+
+
+
+
+class ml.rl.models.example_sequence_model. ExampleSequenceModelOutput ( value : torch.Tensor )
+Bases: object
+
+
+value : torch.Tensor
+
+
+
+
+
+
+ml.rl.models.fully_connected_network module
+
+
+class ml.rl.models.fully_connected_network. FullyConnectedNetwork ( * args : Any , ** kwargs : Any )
+Bases: torch.nn.Module
+
+
+forward ( input : torch.FloatTensor ) → torch.FloatTensor
+Forward pass for generic feed-forward DNNs. Assumes activation names
+are valid pytorch activation names.
+:param input tensor
+
+
+
+
+
+
+ml.rl.models.fully_connected_network. gaussian_fill_w_gain ( tensor , activation , dim_in , min_std = 0.0 ) → None
+Gaussian initialization with gain.
+
+
+
+
+ml.rl.models.mdn_rnn module
+
+
+class ml.rl.models.mdn_rnn. MDNRNN ( * args : Any , ** kwargs : Any )
+Bases: ml.rl.models.mdn_rnn._MDNRNNBase
+Mixture Density Network - Recurrent Neural Network
+
+
+forward ( actions , states , hidden = None )
+Forward pass of MDN-RNN
+
+Parameters
+
+actions – (SEQ_LEN, BATCH_SIZE, ACTION_DIM) torch tensor
+states – (SEQ_LEN, BATCH_SIZE, STATE_DIM) torch tensor
+
+
+Returns
+parameters of the GMM prediction for the next state,
+
+
+gaussian prediction of the reward and logit prediction of
+non-terminality. And the RNN’s outputs.
+
+
+mus: (SEQ_LEN, BATCH_SIZE, NUM_GAUSSIANS, STATE_DIM) torch tensor
+sigmas: (SEQ_LEN, BATCH_SIZE, NUM_GAUSSIANS, STATE_DIM) torch tensor
+logpi: (SEQ_LEN, BATCH_SIZE, NUM_GAUSSIANS) torch tensor
+reward: (SEQ_LEN, BATCH_SIZE) torch tensor
+not_terminal: (SEQ_LEN, BATCH_SIZE) torch tensor
+
+last_step_hidden_and_cell: TUPLE( (NUM_LAYERS, BATCH_SIZE, HIDDEN_SIZE),
+(NUM_LAYERS, BATCH_SIZE, HIDDEN_SIZE)
+
+
+
+
+
) torch tensor
+- all_steps_hidden: (SEQ_LEN, BATCH_SIZE, HIDDEN_SIZE) torch tensor
+
+
+
+
+
+get_initial_hidden_state ( batch_size = 1 )
+
+
+
+
+
+
+class ml.rl.models.mdn_rnn. MDNRNNMemoryPool ( max_replay_memory_size )
+Bases: object
+
+
+deque_sample ( indices )
+
+
+
+
+insert_into_memory ( state , action , next_state , reward , not_terminal )
+
+
+
+
+property memory_size
+
+
+
+
+sample_memories ( batch_size , use_gpu = False , batch_first = False ) → ml.rl.types.PreprocessedTrainingBatch
+
+Parameters
+
+batch_size – number of samples to return
+use_gpu – whether to put samples on gpu
+batch_first – If True, the first dimension of data is batch_size.
+If False (default), the first dimension is SEQ_LEN. Therefore,
+state’s shape is SEQ_LEN x BATCH_SIZE x STATE_DIM, for example. By default,
+MDN-RNN consumes data with SEQ_LEN as the first dimension.
+
+
+
+
+
+
+
+
+
+class ml.rl.models.mdn_rnn. MDNRNNMemorySample ( state , action , next_state , reward , not_terminal )
+Bases: NamedTuple
+
+
+action : numpy.ndarray
+Alias for field number 1
+
+
+
+
+next_state : numpy.ndarray
+Alias for field number 2
+
+
+
+
+not_terminal : float
+Alias for field number 4
+
+
+
+
+reward : float
+Alias for field number 3
+
+
+
+
+state : numpy.ndarray
+Alias for field number 0
+
+
+
+
+
+
+ml.rl.models.mdn_rnn. gmm_loss ( batch , mus , sigmas , logpi , reduce = True )
+Computes the gmm loss.
+Compute minus the log probability of batch under the GMM model described
+by mus, sigmas, pi. Precisely, with bs1, bs2, … the sizes of the batch
+dimensions (several batch dimension are useful when you have both a batch
+axis and a time step axis), gs the number of mixtures and fs the number of
+features.
+
+Parameters
+
+batch – (bs1, bs2, * , fs) torch tensor
+mus – (bs1, bs2, * , gs, fs) torch tensor
+sigmas – (bs1, bs2, * , gs, fs) torch tensor
+logpi – (bs1, bs2, * , gs) torch tensor
+reduce – if not reduce, the mean in the following formula is omitted
+
+
+Returns
+
+
+
+
+loss(batch) = - mean_{i1=0..bs1, i2=0..bs2, …} log(
+sum_{k=1..gs} pi[i1, i2, …, k] * N( batch[i1, i2, …, :] | mus[i1, i2, …, k, :], sigmas[i1, i2, …, k, :]))
+
+
+
+
+NOTE: The loss is not reduced along the feature dimension (i.e. it should
+scale linearily with fs).
+Adapted from: https://github.com/ctallec/world-models
+
+
+
+
+ml.rl.models.mdn_rnn. transpose ( * args )
+
+
+
+
+ml.rl.models.no_soft_update_embedding module
+
+
+class ml.rl.models.no_soft_update_embedding. NoSoftUpdateEmbedding ( num_embeddings : int , embedding_dim : int , padding_idx : Optional [ int ] = None , max_norm : Optional [ float ] = None , norm_type : float = 2.0 , scale_grad_by_freq : bool = False , sparse : bool = False , _weight : Optional [ torch.Tensor ] = None , device = None , dtype = None )
+Bases: torch.nn.modules.sparse.Embedding
+Use this instead of vanilla Embedding module to avoid soft-updating the embedding
+table in the target network.
+
+
+embedding_dim : int
+
+
+
+
+max_norm : Optional [ float ]
+
+
+
+
+norm_type : float
+
+
+
+
+num_embeddings : int
+
+
+
+
+padding_idx : Optional [ int ]
+
+
+
+
+scale_grad_by_freq : bool
+
+
+
+
+sparse : bool
+
+
+
+
+weight : torch.Tensor
+
+
+
+
+
+
+ml.rl.models.parametric_dqn module
+
+
+class ml.rl.models.parametric_dqn. FullyConnectedParametricDQN ( * args : Any , ** kwargs : Any )
+Bases: ml.rl.models.base.ModelBase
+
+
+forward ( input )
+
+
+
+
+get_distributed_data_parallel_model ( )
+Return DistributedDataParallel version of this model
+This needs to be implemented explicitly because:
+1) Model with EmbeddingBag module is not compatible with vanilla DistributedDataParallel
+2) Exporting logic needs structured data. DistributedDataParallel doesn’t work with structured data.
+
+
+
+
+input_prototype ( )
+This function provides the input for ONNX graph tracing.
+The return value should be what expected by forward() .
+
+
+
+
+
+
+class ml.rl.models.parametric_dqn. ParametricDQNWithPreprocessing ( * args : Any , ** kwargs : Any )
+Bases: ml.rl.models.base.ModelBase
+
+
+forward ( input )
+
+
+
+
+input_prototype ( )
+This function provides the input for ONNX graph tracing.
+The return value should be what expected by forward() .
+
+
+
+
+
+
+
+ml.rl.models.seq2slate module
+
+
+class ml.rl.models.seq2slate. BaselineNet ( * args : Any , ** kwargs : Any )
+Bases: torch.nn.Module
+
+
+forward ( input : ml.rl.types.PreprocessedRankingInput )
+
+
+
+
+
+
+class ml.rl.models.seq2slate. Decoder ( * args : Any , ** kwargs : Any )
+Bases: torch.nn.Module
+Generic num_layers layer decoder with masking.
+
+
+forward ( x , memory , tgt_src_mask , tgt_tgt_mask )
+
+
+
+
+
+
+class ml.rl.models.seq2slate. DecoderLayer ( * args : Any , ** kwargs : Any )
+Bases: torch.nn.Module
+Decoder is made of self-attn, src-attn, and feed forward
+
+
+forward ( x , m , tgt_src_mask , tgt_tgt_mask )
+
+
+
+
+
+
+class ml.rl.models.seq2slate. Embedder ( * args : Any , ** kwargs : Any )
+Bases: torch.nn.Module
+
+
+forward ( x )
+
+
+
+
+
+
+class ml.rl.models.seq2slate. Encoder ( * args : Any , ** kwargs : Any )
+Bases: torch.nn.Module
+Core encoder is a stack of num_layers layers
+
+
+forward ( x , mask )
+Pass the input (and mask) through each layer in turn.
+
+
+
+
+
+
+class ml.rl.models.seq2slate. EncoderLayer ( * args : Any , ** kwargs : Any )
+Bases: torch.nn.Module
+Encoder is made up of self-attn and feed forward
+
+
+forward ( src_embed , src_mask )
+
+
+
+
+
+
+class ml.rl.models.seq2slate. Generator ( * args : Any , ** kwargs : Any )
+Bases: torch.nn.Module
+Define standard linear + softmax generation step.
+
+
+forward ( mode , decoder_output = None , tgt_in_idx = None , greedy = None )
+
+
+
+
+
+
+class ml.rl.models.seq2slate. MultiHeadedAttention ( * args : Any , ** kwargs : Any )
+Bases: torch.nn.Module
+
+
+forward ( query , key , value , mask = None )
+
+
+
+
+
+
+class ml.rl.models.seq2slate. PositionalEncoding ( * args : Any , ** kwargs : Any )
+Bases: torch.nn.Module
+
+
+forward ( x , seq_len )
+
+
+
+
+
+
+class ml.rl.models.seq2slate. PositionwiseFeedForward ( * args : Any , ** kwargs : Any )
+Bases: torch.nn.Module
+
+
+forward ( x )
+
+
+
+
+
+
+class ml.rl.models.seq2slate. Seq2SlateMode ( value )
+Bases: enum.Enum
+An enumeration.
+
+
+DECODE_ONE_STEP_MODE = 'decode_one_step'
+
+
+
+
+PER_SEQ_LOG_PROB_MODE = 'per_sequence_log_prob'
+
+
+
+
+PER_SYMBOL_LOG_PROB_DIST_MODE = 'per_symbol_log_prob_dist'
+
+
+
+
+RANK_MODE = 'rank'
+
+
+
+
+
+
+class ml.rl.models.seq2slate. Seq2SlateTransformerModel ( * args : Any , ** kwargs : Any )
+Bases: torch.nn.Module
+A Seq2Slate network with Transformer. The network is essentially an
+encoder-decoder structure. The encoder inputs a sequence of candidate feature
+vectors and a state feature vector, and the decoder outputs an ordered
+list of candidate indices. The output order is learned through REINFORCE
+algorithm to optimize some sequence-wise reward which is also specific to
+the provided state feature.
+One application example is to rank candidate feeds to a specific user such
+that the final list of feeds as a whole optimizes the user’s engagement.
+Seq2Slate paper: https://arxiv.org/abs/1810.02019
+Transformer paper: https://arxiv.org/abs/1706.03762
+
+
+decode ( memory , state , tgt_src_mask , tgt_in_seq , tgt_tgt_mask , tgt_seq_len )
+
+
+
+
+encode ( state , src_seq , src_mask )
+
+
+
+
+forward ( input : ml.rl.types.PreprocessedRankingInput , mode : str , tgt_seq_len : Optional [ int ] = None , greedy : Optional [ bool ] = None )
+
+Parameters
+
+input – model input
+mode – a string indicating which mode to perform.
+“rank”: return ranked actions and their generative probabilities.
+“log_probs”: return generative log probabilities of given tgt sequences
+(used for REINFORCE training)
+tgt_seq_len – the length of output sequence to be decoded. Only used
+in rank mode
+greedy – whether to sample based on softmax distribution or greedily
+when decoding. Only used in rank mode
+
+
+
+
+
+
+
+
+
+class ml.rl.models.seq2slate. Seq2SlateTransformerNet ( * args : Any , ** kwargs : Any )
+Bases: ml.rl.models.base.ModelBase
+
+
+forward ( input : ml.rl.types.PreprocessedRankingInput , mode : str , tgt_seq_len : Optional [ int ] = None , greedy : Optional [ bool ] = None )
+
+
+
+
+get_distributed_data_parallel_model ( )
+Return DistributedDataParallel version of this model
+This needs to be implemented explicitly because:
+1) Model with EmbeddingBag module is not compatible with vanilla DistributedDataParallel
+2) Exporting logic needs structured data. DistributedDataParallel doesn’t work with structured data.
+
+
+
+
+input_prototype ( )
+This function provides the input for ONNX graph tracing.
+The return value should be what expected by forward() .
+
+
+
+
+
+
+class ml.rl.models.seq2slate. SublayerConnection ( * args : Any , ** kwargs : Any )
+Bases: torch.nn.Module
+A residual connection followed by a layer norm.
+
+
+forward ( x , sublayer )
+
+
+
+
+
+
+ml.rl.models.seq2slate. attention ( query , key , value , mask , d_k )
+Scaled Dot Product Attention
+
+
+
+
+ml.rl.models.seq2slate. clones ( module , N )
+Produce N identical layers.
+
+Parameters
+
+module – nn.Module class
+N – number of copies
+
+
+
+
+
+
+
+ml.rl.models.seq2slate. subsequent_and_padding_mask ( tgt_in_idx )
+Create a mask to hide padding and future items
+
+
+
+
+ml.rl.models.seq2slate. subsequent_mask ( size , device )
+Mask out subsequent positions. Mainly used in the decoding process,
+in which an item should not attend subsequent items.
+
+
+
+
+ml.rl.models.seq2slate_reward module
+
+
+class ml.rl.models.seq2slate_reward. Seq2SlateRewardNet ( * args : Any , ** kwargs : Any )
+Bases: ml.rl.models.base.ModelBase
+
+
+decode ( memory , state , tgt_src_mask , tgt_in_seq , tgt_tgt_mask )
+One step decoder. The decoder’s output will be used as the input to
+the last layer for predicting slate reward
+
+
+
+
+encode ( state , src_seq , src_mask )
+
+
+
+
+forward ( input : ml.rl.types.PreprocessedRankingInput )
+Encode tgt sequences and predict the slate reward.
+
+
+
+
+input_prototype ( )
+This function provides the input for ONNX graph tracing.
+The return value should be what expected by forward() .
+
+
+
+
+
+
+class ml.rl.models.seq2slate_reward. Seq2SlateRewardNetJITWrapper ( * args : Any , ** kwargs : Any )
+Bases: ml.rl.models.base.ModelBase
+
+
+forward ( state : torch.Tensor , src_seq : torch.Tensor , tgt_out_seq : torch.Tensor , src_src_mask : torch.Tensor , slate_reward : torch.Tensor , tgt_out_idx : torch.Tensor ) → torch.Tensor
+
+
+
+
+input_prototype ( use_gpu = False )
+This function provides the input for ONNX graph tracing.
+The return value should be what expected by forward() .
+
+
+
+
+
+
+ml.rl.models.world_model module
+
+
+class ml.rl.models.world_model. MemoryNetwork ( * args : Any , ** kwargs : Any )
+Bases: ml.rl.models.base.ModelBase
+
+
+forward ( input )
+
+
+
+
+get_distributed_data_parallel_model ( )
+Return DistributedDataParallel version of this model
+This needs to be implemented explicitly because:
+1) Model with EmbeddingBag module is not compatible with vanilla DistributedDataParallel
+2) Exporting logic needs structured data. DistributedDataParallel doesn’t work with structured data.
+
+
+
+
+input_prototype ( )
+This function provides the input for ONNX graph tracing.
+The return value should be what expected by forward() .
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
\ No newline at end of file
diff --git a/api/ml.rl.polyfill.html b/api/ml.rl.polyfill.html
new file mode 100644
index 00000000..dd3495bd
--- /dev/null
+++ b/api/ml.rl.polyfill.html
@@ -0,0 +1,188 @@
+
+
+
+
+
+
+ ml.rl.polyfill package — ReAgent 1.0 documentation
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+ ReAgent
+
+
+
+
+
+
+
+
+
+ml.rl.polyfill package
+
+
+
+
+ml.rl.polyfill.types module
+Polyfills fblearner types
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
\ No newline at end of file
diff --git a/api/ml.rl.prediction.html b/api/ml.rl.prediction.html
new file mode 100644
index 00000000..44b60638
--- /dev/null
+++ b/api/ml.rl.prediction.html
@@ -0,0 +1,146 @@
+
+
+
+
+
+
+ ml.rl.prediction package — ReAgent 1.0 documentation
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+ ReAgent
+
+
+
+
+
+
+
+
+
+ml.rl.prediction package
+
+
+ml.rl.prediction.dqn_torch_predictor module
+
+
+ml.rl.prediction.predictor_wrapper module
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
\ No newline at end of file
diff --git a/api/ml.rl.preprocessing.html b/api/ml.rl.preprocessing.html
new file mode 100644
index 00000000..9fe3fd17
--- /dev/null
+++ b/api/ml.rl.preprocessing.html
@@ -0,0 +1,167 @@
+
+
+
+
+
+
+ ml.rl.preprocessing package — ReAgent 1.0 documentation
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+ ReAgent
+
+
+
+
+
+
+
+
+
+ml.rl.preprocessing package
+
+
+ml.rl.preprocessing.batch_preprocessor module
+
+
+ml.rl.preprocessing.identify_types module
+
+
+ml.rl.preprocessing.identify_types. identify_type ( values , enum_threshold = 100 )
+
+
+
+
+ml.rl.preprocessing.normalization module
+
+
+ml.rl.preprocessing.postprocessor module
+
+
+ml.rl.preprocessing.preprocessor module
+
+
+ml.rl.preprocessing.sparse_to_dense module
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
\ No newline at end of file
diff --git a/api/ml.rl.readers.html b/api/ml.rl.readers.html
new file mode 100644
index 00000000..ba760b38
--- /dev/null
+++ b/api/ml.rl.readers.html
@@ -0,0 +1,312 @@
+
+
+
+
+
+
+ ml.rl.readers package — ReAgent 1.0 documentation
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+ ReAgent
+
+
+
+
+
+
+
+
+
+ml.rl.readers package
+
+
+ml.rl.readers.base module
+
+
+class ml.rl.readers.base. ReaderBase ( batch_size = None , drop_small = True , num_shards = None )
+Bases: object
+
+
+do_get_shard ( shard_id : int )
+Subclass should implement this if the reader is shardable
+
+
+
+
+get_shard ( shard_id : int )
+Returns a shard of this reader
+
+
+
+
+
+
+class ml.rl.readers.base. ReaderIter
+Bases: object
+
+
+abstract read_batch ( ) → Optional [ collections.OrderedDict ]
+Read a batch of data. The return value should be an OrderedDict.
+Returns None when there is no more data.
+
+
+
+
+
+
+ml.rl.readers.data_streamer module
+
+
+class ml.rl.readers.data_streamer. DataStreamer ( data_reader , num_workers = 0 , pin_memory = False , timeout = 0 , worker_init_fn = None )
+Bases: object
+Data streamer. Provides single- or multi-process iterators over the data_reader.
+
+Parameters
+
+data_reader (DataReader ) – data_reader from which to stream the data.
+num_workers (int , optional ) – how many subprocesses to use for data
+loading. 0 means that the data will be loaded in the main process.
+(default: 0)
+pin_memory (bool , optional ) – If True , the data streamer will copy tensors
+into CUDA pinned memory before returning them.
+timeout (numeric , optional ) – if positive, the timeout value for collecting a
+batch from workers. Should always be non-negative. (default: 0)
+worker_init_fn (callable , optional ) – If not None, this will be called on each
+worker subprocess with the worker id (an int in [0, num_workers - 1] ) as
+input, after seeding and before data loading. (default: None)
+
+
+
+
+
Note
+
By default, each worker will have its PyTorch seed set to
+base_seed + worker_id , where base_seed is a long generated
+by main process using its RNG. However, seeds for other libraries
+may be duplicated upon initializing workers (w.g., NumPy), causing
+each worker to return identical random numbers. (See
+datastreamer-workers-random-seed section in FAQ.) You may
+use torch.initial_seed() to access the PyTorch seed for each
+worker in worker_init_fn , and use it to set other seeds
+before data loading.
+
+
+
Warning
+
If spawn start method is used, worker_init_fn cannot be an
+unpickleable object, e.g., a lambda function.
+
+
+
+
+
+class ml.rl.readers.data_streamer. WorkerDone ( worker_id )
+Bases: tuple
+
+
+worker_id
+Alias for field number 0
+
+
+
+
+
+
+ml.rl.readers.data_streamer. pin_memory ( batch )
+This is ripped off from dataloader. The only difference is that it preserves
+the type of Mapping so that the OrderedDict is maintained.
+
+
+
+
+ml.rl.readers.json_dataset_reader module
+
+
+class ml.rl.readers.json_dataset_reader. JSONDatasetReader ( path , batch_size = None , preprocess_handler = None )
+Bases: ml.rl.readers.base.ReaderBase
+Create the reader for a JSON training dataset.
+
+
+line_count ( )
+
+
+
+
+read_all ( )
+
+
+
+
+read_batch ( )
+
+
+
+
+reset_iterator ( )
+
+
+
+
+
+
+class ml.rl.readers.json_dataset_reader. JSONDatasetReaderIter ( reader )
+Bases: ml.rl.readers.base.ReaderIter
+
+
+read_batch ( ) → Optional [ collections.OrderedDict ]
+Read a batch of data. The return value should be an OrderedDict.
+Returns None when there is no more data.
+
+
+
+
+
+
+ml.rl.readers.nparray_reader module
+
+
+class ml.rl.readers.nparray_reader. NpArrayReader ( data , size = None , ** kwargs )
+Bases: ml.rl.readers.base.ReaderBase
+Basic reader taking np.ndarray`s of a whole dataset and split them into
+chunks of `batch_size .
+
+
+do_get_shard ( shard_id : int )
+Subclass should implement this if the reader is shardable
+
+
+
+
+
+
+class ml.rl.readers.nparray_reader. NpArrayReaderIter ( reader )
+Bases: ml.rl.readers.base.ReaderIter
+
+
+read_batch ( ) → Optional [ collections.OrderedDict ]
+Read a batch of data. The return value should be an OrderedDict.
+Returns None when there is no more data.
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
\ No newline at end of file
diff --git a/api/ml.rl.simulators.html b/api/ml.rl.simulators.html
new file mode 100644
index 00000000..f794fd2e
--- /dev/null
+++ b/api/ml.rl.simulators.html
@@ -0,0 +1,273 @@
+
+
+
+
+
+
+ ml.rl.simulators package — ReAgent 1.0 documentation
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+ ReAgent
+
+
+
+
+
+
+
+
+
+ml.rl.simulators package
+
+
+ml.rl.simulators.recsim module
+
+
+class ml.rl.simulators.recsim. DocumentFeature ( topic , length , quality )
+Bases: NamedTuple
+
+
+as_vector ( )
+Convenient function to get single tensor
+
+
+
+
+length : torch.Tensor
+Alias for field number 1
+
+
+
+
+quality : torch.Tensor
+Alias for field number 2
+
+
+
+
+topic : torch.Tensor
+Alias for field number 0
+
+
+
+
+
+
+class ml.rl.simulators.recsim. RecSim ( num_topics : int = 20 , doc_length : float = 4 , quality_means : List [ Tuple [ float , float ] ] = [(- 3.0, 0.0), (- 3.0, 0.0), (- 3.0, 0.0), (- 3.0, 0.0), (- 3.0, 0.0), (- 3.0, 0.0), (- 3.0, 0.0), (- 3.0, 0.0), (- 3.0, 0.0), (- 3.0, 0.0), (- 3.0, 0.0), (- 3.0, 0.0), (- 3.0, 0.0), (- 3.0, 0.0), (0.0, 3.0), (0.0, 3.0), (0.0, 3.0), (0.0, 3.0), (0.0, 3.0), (0.0, 3.0)] , quality_variances : List [ float ] = [1.0, 1.0, 1.0, 1.0, 1.0, 1.0, 1.0, 1.0, 1.0, 1.0, 1.0, 1.0, 1.0, 1.0, 1.0, 1.0, 1.0, 1.0, 1.0, 1.0] , initial_budget : float = 200 , alpha : float = 1.0 , m : int = 10 , k : int = 3 , num_users : int = 5000 , y : float = 0.3 , device : str = 'cpu' , seed : int = 2147483647 )
+Bases: object
+An environment described in Section 6 of https://arxiv.org/abs/1905.12767
+
+
+bonus ( u , d , length , quality )
+
+
+
+
+compute_user_choice ( slate : ml.rl.simulators.recsim.DocumentFeature ) → Tuple [ torch.Tensor , torch.Tensor ]
+
+
+
+
+interest ( u , d )
+
+Parameters
+
+u – shape [batch, T]
+d – shape [batch, k, T]
+
+
+
+
+
+
+
+obs ( ) → Tuple [ torch.Tensor , torch.Tensor , ml.rl.simulators.recsim.DocumentFeature ]
+Agent can observe:
+- User interest vector
+- Document topic vector
+- Document length
+- Document quality
+
+
+
+
+reset ( ) → None
+
+
+
+
+rollout_policy ( policy , memory_pool : Optional [ ml.rl.test.gym.open_ai_gym_memory_pool.OpenAIGymMemoryPool ] = None ) → float
+
+
+
+
+sample_documents ( n : int ) → ml.rl.simulators.recsim.DocumentFeature
+
+
+
+
+sample_users ( n )
+User is represented by vector of topic interest, uniformly sampled from [-1, 1]
+
+
+
+
+satisfactory ( u , d , quality )
+
+
+
+
+select ( candidates : ml.rl.simulators.recsim.DocumentFeature , indices : torch.Tensor , add_null : bool ) → ml.rl.simulators.recsim.DocumentFeature
+
+
+
+
+step ( action : torch.Tensor ) → Tuple [ torch.Tensor , torch.Tensor , torch.Tensor , int ]
+
+
+
+
+update_active_users ( ) → int
+
+
+
+
+update_user_budget ( selected_choice )
+
+
+
+
+update_user_interest ( selected_choice )
+
+
+
+
+
+
+ml.rl.simulators.recsim. random_policy ( obs : Tuple [ torch.Tensor , torch.Tensor , ml.rl.simulators.recsim.DocumentFeature ] , recsim : ml.rl.simulators.recsim.RecSim )
+
+
+
+
+ml.rl.simulators.recsim. top_k_policy ( q_network , obs : Tuple [ torch.Tensor , torch.Tensor , ml.rl.simulators.recsim.DocumentFeature ] , recsim : ml.rl.simulators.recsim.RecSim )
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
\ No newline at end of file
diff --git a/api/ml.rl.training.gradient_free.html b/api/ml.rl.training.gradient_free.html
new file mode 100644
index 00000000..a1887b70
--- /dev/null
+++ b/api/ml.rl.training.gradient_free.html
@@ -0,0 +1,174 @@
+
+
+
+
+
+
+ ml.rl.training.gradient_free package — ReAgent 1.0 documentation
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+ ReAgent
+
+
+
+
+
+
+
+
+
+ml.rl.training.gradient_free package
+
+
+ml.rl.training.gradient_free.es_worker module
+
+
+ml.rl.training.gradient_free.evolution_pool module
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
\ No newline at end of file
diff --git a/api/ml.rl.training.html b/api/ml.rl.training.html
new file mode 100644
index 00000000..b1765db8
--- /dev/null
+++ b/api/ml.rl.training.html
@@ -0,0 +1,780 @@
+
+
+
+
+
+
+ ml.rl.training package — ReAgent 1.0 documentation
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+ ReAgent
+
+
+
+
+
+
+
+
+
+ml.rl.training package
+
+
+
+ml.rl.training.c51_trainer module
+
+
+ml.rl.training.cem_trainer module
+
+
+ml.rl.training.dqn_trainer module
+
+
+ml.rl.training.dqn_trainer_base module
+
+
+ml.rl.training.imitator_training module
+
+
+ml.rl.training.loss_reporter module
+
+
+class ml.rl.training.loss_reporter. BatchStats ( td_loss , reward_loss , imitator_loss , logged_actions , logged_propensities , logged_rewards , logged_values , model_propensities , model_rewards , model_values , model_values_on_logged_actions , model_action_idxs )
+Bases: NamedTuple
+
+
+static add_custom_scalars ( action_names : Optional [ List [ str ] ] )
+
+
+
+
+imitator_loss : Optional [ torch.Tensor ]
+Alias for field number 2
+
+
+
+
+logged_actions : Optional [ torch.Tensor ]
+Alias for field number 3
+
+
+
+
+logged_propensities : Optional [ torch.Tensor ]
+Alias for field number 4
+
+
+
+
+logged_rewards : Optional [ torch.Tensor ]
+Alias for field number 5
+
+
+
+
+logged_values : Optional [ torch.Tensor ]
+Alias for field number 6
+
+
+
+
+model_action_idxs : Optional [ torch.Tensor ]
+Alias for field number 11
+
+
+
+
+model_propensities : Optional [ torch.Tensor ]
+Alias for field number 7
+
+
+
+
+model_rewards : Optional [ torch.Tensor ]
+Alias for field number 8
+
+
+
+
+model_values : Optional [ torch.Tensor ]
+Alias for field number 9
+
+
+
+
+model_values_on_logged_actions : Optional [ torch.Tensor ]
+Alias for field number 10
+
+
+
+
+reward_loss : Optional [ torch.Tensor ]
+Alias for field number 1
+
+
+
+
+td_loss : Optional [ torch.Tensor ]
+Alias for field number 0
+
+
+
+
+write_summary ( actions : List [ str ] )
+
+
+
+
+
+
+class ml.rl.training.loss_reporter. LossReporter ( action_names : Optional [ List [ str ] ] = None )
+Bases: object
+
+
+RECENT_WINDOW_SIZE = 100
+
+
+
+
+static calculate_recent_window_average ( arr , window_size , num_entries )
+
+
+
+
+flush ( )
+
+
+
+
+get_logged_action_distribution ( )
+
+
+
+
+get_model_action_distribution ( )
+
+
+
+
+get_recent_imitator_loss ( )
+
+
+
+
+get_recent_reward_loss ( )
+
+
+
+
+get_recent_rewards ( )
+
+
+
+
+get_recent_td_loss ( )
+
+
+
+
+get_td_loss_after_n ( n )
+
+
+
+
+log_to_tensorboard ( epoch : int ) → None
+
+
+
+
+property num_batches
+
+
+
+
+report ( ** kwargs )
+
+
+
+
+
+
+class ml.rl.training.loss_reporter. StatsByAction ( actions )
+Bases: object
+
+
+append ( stats )
+
+
+
+
+items ( )
+
+
+
+
+
+
+ml.rl.training.loss_reporter. merge_tensor_namedtuple_list ( l , cls )
+
+
+
+
+ml.rl.training.on_policy_predictor module
+
+
+class ml.rl.training.on_policy_predictor. CEMPlanningPredictor ( trainer , action_dim : int , use_gpu : bool )
+Bases: ml.rl.training.on_policy_predictor.OnPolicyPredictor
+
+
+discrete_action ( ) → bool
+Return True if this predictor is for a discrete action network
+
+
+
+
+policy ( states : torch.Tensor , possible_actions_presence : Optional [ torch.Tensor ] = None ) → Union [ ml.rl.types.SacPolicyActionSet , ml.rl.types.DqnPolicyActionSet ]
+
+
+
+
+policy_net ( ) → bool
+Return True if this predictor is for a policy network
+
+
+
+
+
+
+class ml.rl.training.on_policy_predictor. ContinuousActionOnPolicyPredictor ( trainer , action_dim : int , use_gpu : bool )
+Bases: ml.rl.training.on_policy_predictor.OnPolicyPredictor
+
+
+policy ( states : torch.Tensor ) → ml.rl.types.SacPolicyActionSet
+
+
+
+
+policy_net ( ) → bool
+Return True if this predictor is for a policy network
+
+
+
+
+
+
+class ml.rl.training.on_policy_predictor. DiscreteDQNOnPolicyPredictor ( trainer , action_dim : int , use_gpu : bool )
+Bases: ml.rl.training.on_policy_predictor.OnPolicyPredictor
+
+
+discrete_action ( ) → bool
+Return True if this predictor is for a discrete action network
+
+
+
+
+estimate_reward ( state )
+
+
+
+
+policy ( state : torch.Tensor , possible_actions_presence : torch.Tensor ) → ml.rl.types.DqnPolicyActionSet
+
+
+
+
+policy_net ( ) → bool
+Return True if this predictor is for a policy network
+
+
+
+
+predict ( state )
+
+
+
+
+
+
+class ml.rl.training.on_policy_predictor. OnPolicyPredictor ( trainer , action_dim : int , use_gpu : bool )
+Bases: object
+This class generates actions given a trainer and a state. It’s used for
+on-policy learning. If you have a TorchScript (i.e. serialized) model,
+Use the classes in off_policy_predictor.py
+
+
+discrete_action ( ) → bool
+Return True if this predictor is for a discrete action network
+
+
+
+
+policy_net ( ) → bool
+Return True if this predictor is for a policy network
+
+
+
+
+
+
+class ml.rl.training.on_policy_predictor. ParametricDQNOnPolicyPredictor ( trainer , action_dim : int , use_gpu : bool )
+Bases: ml.rl.training.on_policy_predictor.OnPolicyPredictor
+
+
+discrete_action ( ) → bool
+Return True if this predictor is for a discrete action network
+
+
+
+
+estimate_reward ( states_tiled : torch.Tensor , possible_actions : torch.Tensor )
+
+
+
+
+policy ( states_tiled : torch.Tensor , possible_actions_with_presence : Tuple [ torch.Tensor , torch.Tensor ] )
+
+
+
+
+policy_net ( ) → bool
+Return True if this predictor is for a policy network
+
+
+
+
+predict ( states_tiled : torch.Tensor , possible_actions : torch.Tensor )
+
+
+
+
+
+
+ml.rl.training.parametric_dqn_trainer module
+
+
+ml.rl.training.qrdqn_trainer module
+
+
+
+ml.rl.training.rl_dataset module
+
+
+class ml.rl.training.rl_dataset. RLDataset ( file_path )
+Bases: object
+
+
+insert ( ** kwargs )
+
+
+
+
+insert_pre_timeline_format ( mdp_id , sequence_number , state , timeline_format_action , reward , possible_actions , time_diff , action_probability , possible_actions_mask )
+Insert a new sample to the dataset in the pre-timeline json format.
+Format needed for running timeline operator and for uploading dataset to hive.
+
+
+
+
+insert_replay_buffer_format ( state , action , reward , next_state , next_action , terminal , possible_next_actions , possible_next_actions_mask , time_diff , possible_actions , possible_actions_mask , policy_id )
+Insert a new sample to the dataset in the same format as the
+replay buffer.
+
+
+
+
+load ( )
+Load samples from a gzipped json file.
+
+
+
+
+save ( )
+Save samples as a pickle file or JSON file.
+
+
+
+
+
+
+ml.rl.training.rl_trainer_pytorch module
+
+
+ml.rl.training.sac_trainer module
+
+
+ml.rl.training.slate_q_trainer module
+
+
+ml.rl.training.td3_trainer module
+
+
+ml.rl.training.trainer module
+
+
+class ml.rl.training.trainer. Trainer
+Bases: object
+
+
+load_state_dict ( state_dict )
+
+
+
+
+state_dict ( )
+
+
+
+
+train ( training_batch : ml.rl.types.PreprocessedTrainingBatch ) → None
+
+
+
+
+warm_start_components ( ) → List [ str ]
+The trainer should specify what members to save and load
+
+
+
+
+
+
+ml.rl.training.training_data_page module
+
+
+class ml.rl.training.training_data_page. TrainingDataPage ( mdp_ids : Optional [ numpy.ndarray ] = None , sequence_numbers : Optional [ torch.Tensor ] = None , states : Optional [ torch.Tensor ] = None , actions : Optional [ torch.Tensor ] = None , propensities : Optional [ torch.Tensor ] = None , rewards : Optional [ torch.Tensor ] = None , possible_actions_mask : Optional [ torch.Tensor ] = None , possible_actions_state_concat : Optional [ torch.Tensor ] = None , next_states : Optional [ torch.Tensor ] = None , next_actions : Optional [ torch.Tensor ] = None , possible_next_actions_mask : Optional [ torch.Tensor ] = None , possible_next_actions_state_concat : Optional [ torch.Tensor ] = None , not_terminal : Optional [ torch.Tensor ] = None , time_diffs : Optional [ torch.Tensor ] = None , metrics : Optional [ torch.Tensor ] = None , step : Optional [ torch.Tensor ] = None , max_num_actions : Optional [ int ] = None , next_propensities : Optional [ torch.Tensor ] = None , rewards_mask : Optional [ torch.Tensor ] = None )
+Bases: object
+
+
+actions
+
+
+
+
+as_cem_training_batch ( batch_first = False )
+Generate one-step samples needed by CEM trainer.
+The samples will be used to train an ensemble of world models used by CEM.
+
+If batch_first = True: state/next state shape: batch_size x 1 x state_dim
+action shape: batch_size x 1 x action_dim
+reward/terminal shape: batch_size x 1
+
+else (default): state/next state shape: 1 x batch_size x state_dim
+action shape: 1 x batch_size x action_dim
+reward/terminal shape: 1 x batch_size
+
+
+
+
+
+
+as_discrete_maxq_training_batch ( )
+
+
+
+
+as_parametric_maxq_training_batch ( )
+
+
+
+
+as_policy_network_training_batch ( )
+
+
+
+
+as_slate_q_training_batch ( )
+
+
+
+
+max_num_actions
+
+
+
+
+mdp_ids
+
+
+
+
+metrics
+
+
+
+
+next_actions
+
+
+
+
+next_propensities
+
+
+
+
+next_states
+
+
+
+
+not_terminal
+
+
+
+
+possible_actions_mask
+
+
+
+
+possible_actions_state_concat
+
+
+
+
+possible_next_actions_mask
+
+
+
+
+possible_next_actions_state_concat
+
+
+
+
+propensities
+
+
+
+
+rewards
+
+
+
+
+rewards_mask
+
+
+
+
+sequence_numbers
+
+
+
+
+set_device ( device )
+
+
+
+
+set_type ( dtype )
+
+
+
+
+size ( ) → int
+
+
+
+
+states
+
+
+
+
+step
+
+
+
+
+time_diffs
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
\ No newline at end of file
diff --git a/api/ml.rl.training.ranking.html b/api/ml.rl.training.ranking.html
new file mode 100644
index 00000000..4e9bd571
--- /dev/null
+++ b/api/ml.rl.training.ranking.html
@@ -0,0 +1,174 @@
+
+
+
+
+
+
+ ml.rl.training.ranking package — ReAgent 1.0 documentation
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+ ReAgent
+
+
+
+
+
+
+
+
+
+ml.rl.training.ranking package
+
+
+ml.rl.training.ranking.seq2slate_tf_trainer module
+
+
+ml.rl.training.ranking.seq2slate_trainer module
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
\ No newline at end of file
diff --git a/api/ml.rl.training.world_model.html b/api/ml.rl.training.world_model.html
new file mode 100644
index 00000000..ad2fe139
--- /dev/null
+++ b/api/ml.rl.training.world_model.html
@@ -0,0 +1,170 @@
+
+
+
+
+
+
+ ml.rl.training.world_model package — ReAgent 1.0 documentation
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+ ReAgent
+
+
+
+
+
+
+
+
+
+ml.rl.training.world_model package
+
+
+ml.rl.training.world_model.mdnrnn_trainer module
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
\ No newline at end of file
diff --git a/api/ml.rl.workflow.html b/api/ml.rl.workflow.html
new file mode 100644
index 00000000..3f1f6937
--- /dev/null
+++ b/api/ml.rl.workflow.html
@@ -0,0 +1,170 @@
+
+
+
+
+
+
+ ml.rl.workflow package — ReAgent 1.0 documentation
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+ ReAgent
+
+
+
+
+
+
+
+
+
+ml.rl.workflow package
+
+
+ml.rl.workflow.base_workflow module
+
+
+
+ml.rl.workflow.dqn_workflow module
+
+
+ml.rl.workflow.helpers module
+
+
+ml.rl.workflow.page_handler module
+
+
+ml.rl.workflow.parametric_dqn_workflow module
+
+
+ml.rl.workflow.preprocess_handler module
+
+
+ml.rl.workflow.transitional module
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
\ No newline at end of file
diff --git a/api/reagent.training.cb.html b/api/reagent.training.cb.html
new file mode 100644
index 00000000..32e0d9a0
--- /dev/null
+++ b/api/reagent.training.cb.html
@@ -0,0 +1,461 @@
+
+
+
+
+
+
+ reagent.training.cb package — ReAgent 1.0 documentation
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+ ReAgent
+
+
+
+
+
+
+
+
+
+reagent.training.cb package
+
+
+reagent.training.cb.linucb_trainer module
+
+
+class reagent.training.cb.linucb_trainer. LinUCBTrainer ( policy : reagent.gym.policies.policy.Policy , num_actions : int = - 1 , use_interaction_features : bool = True )
+Bases: reagent.training.reagent_lightning_module.ReAgentLightningModule
+The trainer for LinUCB Contextual Bandit model.
+The model estimates a ridge regression (linear) and only supports dense features.
+The actions are assumed to be one of:
+
+
+
+Fixed actions. The same (have the same semantic meaning) actions across all contexts. If actions are fixed, they can’t have features associated with them.
+
+
+
+
+Feature actions. We can have different number and identities of actions in each context. The actions must have features to represent their semantic meaning.
+
+
+
+
+
+Reference: https://arxiv.org/pdf/1003.0146.pdf
+
+Parameters
+
+policy – The policy to be trained. Its scorer has to be LinearRegressionUCB
+num_actions – The number of actions. If num_actions==-1, the actions are assumed to be feature actions,
+otherwise they are assumed to be fixed actions.
+use_interaction_features – If True,
+
+
+
+
+
+allow_zero_length_dataloader_with_multiple_devices : bool
+
+
+
+
+configure_optimizers ( )
+Choose what optimizers and learning-rate schedulers to use in your optimization.
+Normally you’d need one. But in the case of GANs or similar you might have multiple.
+
+Returns
+
Any of these 6 options.
+
+Single optimizer .
+List or Tuple of optimizers.
+Two lists - The first list has multiple optimizers, and the second has multiple LR schedulers
+(or multiple lr_scheduler_config ).
+Dictionary , with an "optimizer" key, and (optionally) a "lr_scheduler"
+key whose value is a single LR scheduler or lr_scheduler_config .
+Tuple of dictionaries as described above, with an optional "frequency" key.
+None - Fit will run without any optimizer.
+
+
+
+
+The lr_scheduler_config is a dictionary which contains the scheduler and its associated configuration.
+The default configuration is shown below.
+lr_scheduler_config = {
+ # REQUIRED: The scheduler instance
+ "scheduler" : lr_scheduler ,
+ # The unit of the scheduler's step size, could also be 'step'.
+ # 'epoch' updates the scheduler on epoch end whereas 'step'
+ # updates it after a optimizer update.
+ "interval" : "epoch" ,
+ # How many epochs/steps should pass between calls to
+ # `scheduler.step()`. 1 corresponds to updating the learning
+ # rate after every epoch/step.
+ "frequency" : 1 ,
+ # Metric to to monitor for schedulers like `ReduceLROnPlateau`
+ "monitor" : "val_loss" ,
+ # If set to `True`, will enforce that the value specified 'monitor'
+ # is available when the scheduler is updated, thus stopping
+ # training if not found. If set to `False`, it will only produce a warning
+ "strict" : True ,
+ # If using the `LearningRateMonitor` callback to monitor the
+ # learning rate progress, this keyword can be used to specify
+ # a custom logged name
+ "name" : None ,
+}
+
+
+When there are schedulers in which the .step() method is conditioned on a value, such as the
+torch.optim.lr_scheduler.ReduceLROnPlateau scheduler, Lightning requires that the
+lr_scheduler_config contains the keyword "monitor" set to the metric name that the scheduler
+should be conditioned on.
+Metrics can be made available to monitor by simply logging it using
+self.log('metric_to_track', metric_val) in your LightningModule .
+
+
Note
+
The frequency value specified in a dict along with the optimizer key is an int corresponding
+to the number of sequential batches optimized with the specific optimizer.
+It should be given to none or to all of the optimizers.
+There is a difference between passing multiple optimizers in a list,
+and passing multiple optimizers in dictionaries with a frequency of 1:
+
+
+In the former case, all optimizers will operate on the given batch in each optimization step.
+In the latter, only one optimizer will operate on the given batch at every step.
+
+
+
This is different from the frequency value specified in the lr_scheduler_config mentioned above.
+
def configure_optimizers ( self ):
+ optimizer_one = torch . optim . SGD ( self . model . parameters (), lr = 0.01 )
+ optimizer_two = torch . optim . SGD ( self . model . parameters (), lr = 0.01 )
+ return [
+ { "optimizer" : optimizer_one , "frequency" : 5 },
+ { "optimizer" : optimizer_two , "frequency" : 10 },
+ ]
+
+
+
In this example, the first optimizer will be used for the first 5 steps,
+the second optimizer for the next 10 steps and that cycle will continue.
+If an LR scheduler is specified for an optimizer using the lr_scheduler key in the above dict,
+the scheduler will only be updated when its optimizer is being used.
+
+Examples:
+# most cases. no learning rate scheduler
+def configure_optimizers ( self ):
+ return Adam ( self . parameters (), lr = 1e-3 )
+
+# multiple optimizer case (e.g.: GAN)
+def configure_optimizers ( self ):
+ gen_opt = Adam ( self . model_gen . parameters (), lr = 0.01 )
+ dis_opt = Adam ( self . model_dis . parameters (), lr = 0.02 )
+ return gen_opt , dis_opt
+
+# example with learning rate schedulers
+def configure_optimizers ( self ):
+ gen_opt = Adam ( self . model_gen . parameters (), lr = 0.01 )
+ dis_opt = Adam ( self . model_dis . parameters (), lr = 0.02 )
+ dis_sch = CosineAnnealing ( dis_opt , T_max = 10 )
+ return [ gen_opt , dis_opt ], [ dis_sch ]
+
+# example with step-based learning rate schedulers
+# each optimizer has its own scheduler
+def configure_optimizers ( self ):
+ gen_opt = Adam ( self . model_gen . parameters (), lr = 0.01 )
+ dis_opt = Adam ( self . model_dis . parameters (), lr = 0.02 )
+ gen_sch = {
+ 'scheduler' : ExponentialLR ( gen_opt , 0.99 ),
+ 'interval' : 'step' # called after each training step
+ }
+ dis_sch = CosineAnnealing ( dis_opt , T_max = 10 ) # called every epoch
+ return [ gen_opt , dis_opt ], [ gen_sch , dis_sch ]
+
+# example with optimizer frequencies
+# see training procedure in `Improved Training of Wasserstein GANs`, Algorithm 1
+# https://arxiv.org/abs/1704.00028
+def configure_optimizers ( self ):
+ gen_opt = Adam ( self . model_gen . parameters (), lr = 0.01 )
+ dis_opt = Adam ( self . model_dis . parameters (), lr = 0.02 )
+ n_critic = 5
+ return (
+ { 'optimizer' : dis_opt , 'frequency' : n_critic },
+ { 'optimizer' : gen_opt , 'frequency' : 1 }
+ )
+
+
+
+
Note
+
Some things to know:
+
+Lightning calls .backward() and .step() on each optimizer and learning rate scheduler as needed.
+If you use 16-bit precision (precision=16 ), Lightning will automatically handle the optimizers.
+If you use multiple optimizers, training_step() will have an additional optimizer_idx parameter.
+If you use torch.optim.LBFGS , Lightning handles the closure function automatically for you.
+If you use multiple optimizers, gradients will be calculated only for the parameters of current optimizer
+at each training step.
+If you need to control how often those optimizers step or override the default .step() schedule,
+override the optimizer_step() hook.
+
+
+
+
+
+
+precision : int
+
+
+
+
+prepare_data_per_node : bool
+
+
+
+
+training : bool
+
+
+
+
+training_step ( batch : reagent.core.types.CBInput , batch_idx : int , optimizer_idx : int = 0 )
+Here you compute and return the training loss and some additional metrics for e.g.
+the progress bar or logger.
+
+Parameters
+
+
+Returns
+
Any of.
+
+
+
+
+In this step you’d normally do the forward pass and calculate the loss for a batch.
+You can also do fancier things like multiple forward passes or something model specific.
+Example:
+def training_step ( self , batch , batch_idx ):
+ x , y , z = batch
+ out = self . encoder ( x )
+ loss = self . loss ( out , x )
+ return loss
+
+
+If you define multiple optimizers, this step will be called with an additional
+optimizer_idx parameter.
+# Multiple optimizers (e.g.: GANs)
+def training_step ( self , batch , batch_idx , optimizer_idx ):
+ if optimizer_idx == 0 :
+ # do training_step with encoder
+ ...
+ if optimizer_idx == 1 :
+ # do training_step with decoder
+ ...
+
+
+If you add truncated back propagation through time you will also get an additional
+argument with the hidden states of the previous step.
+# Truncated back-propagation through time
+def training_step ( self , batch , batch_idx , hiddens ):
+ # hiddens are the hidden states from the previous truncated backprop step
+ out , hiddens = self . lstm ( data , hiddens )
+ loss = ...
+ return { "loss" : loss , "hiddens" : hiddens }
+
+
+
+
Note
+
The loss value shown in the progress bar is smoothed (averaged) over the last values,
+so it differs from the actual loss returned in train/validation step.
+
+
+
+
+
+update_params ( x : torch.Tensor , y : torch.Tensor , weight : Optional [ torch.Tensor ] = None )
+
+Parameters
+
+x – 2D tensor of shape (batch_size, dim)
+y – 2D tensor of shape (batch_size, 1)
+weight – 2D tensor of shape (batch_size, 1)
+
+
+
+
+
+
+
+use_amp : bool
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
\ No newline at end of file