Merge branch 'main' of github.com:sisl/InteractionImitation
This commit is contained in:
9
generate_demos.sh
Executable file
9
generate_demos.sh
Executable file
@@ -0,0 +1,9 @@
|
|||||||
|
#DEFAULT PARAMETERS:
|
||||||
|
# locs:list=None, (default to all locations)
|
||||||
|
# tracks:list=None, (default to all tracks)
|
||||||
|
# env_class:str='NRasterizedIncrementingAgent',
|
||||||
|
# env_args:dict={width:36,height:36,m_per_px:2},
|
||||||
|
# expert_class:str='NormalizedIntersimpleExpert',
|
||||||
|
# expert_args:dict={mu:0.001}):
|
||||||
|
|
||||||
|
python -m src.data.expert --locs='[DR_USA_Roundabout_FT]'
|
||||||
@@ -1,55 +1,40 @@
|
|||||||
# %%
|
# %%
|
||||||
from gail.discriminator import CnnDiscriminator, CnnDiscriminatorFlatAction
|
import sys
|
||||||
|
sys.path.append('../../../')
|
||||||
|
from src.discriminator import CnnDiscriminator, CnnDiscriminatorFlatAction
|
||||||
|
from src.policies import OptionsCnnPolicy
|
||||||
|
from src.util import feasible
|
||||||
|
from src.data import load_experts
|
||||||
|
|
||||||
from imitation.algorithms import adversarial
|
from imitation.algorithms import adversarial
|
||||||
|
from imitation.util import logger
|
||||||
|
import imitation.data.rollout as rollout
|
||||||
|
|
||||||
import stable_baselines3
|
import stable_baselines3
|
||||||
|
from stable_baselines3.common.env_util import make_vec_env
|
||||||
|
|
||||||
|
import torch
|
||||||
import torch.utils.data
|
import torch.utils.data
|
||||||
import numpy as np
|
import numpy as np
|
||||||
from intersim.envs.intersimple import NRasterized
|
|
||||||
from intersim.collisions import state_to_polygon
|
|
||||||
import itertools
|
import itertools
|
||||||
from torch.distributions import Categorical
|
|
||||||
import gym
|
import gym
|
||||||
import torch
|
|
||||||
import pickle
|
import pickle
|
||||||
import imitation.data.rollout as rollout
|
|
||||||
import tempfile
|
import tempfile
|
||||||
import pathlib
|
import pathlib
|
||||||
from imitation.util import logger
|
|
||||||
from stable_baselines3.common.env_util import make_vec_env
|
|
||||||
from tqdm import tqdm
|
from tqdm import tqdm
|
||||||
|
|
||||||
model_name = 'gail_options_image'
|
from intersim.envs.intersimple import NRasterized, NRasterizedRandomAgent, NRasterizedIncrementingAgent
|
||||||
env_settings = {'agent': 51, 'width': 36, 'height': 36, 'm_per_px': 2}
|
|
||||||
|
|
||||||
ALL_OPTIONS = [(v,t) for v in [0,2,4,6,8] for t in [5, 10, 20]] # option 0 is safe fallback
|
ALL_OPTIONS = [(v,t) for v in [0,2,4,6,8] for t in [5, 10, 20]] # option 0 is safe fallback
|
||||||
|
|
||||||
class OptionsCnnPolicy(stable_baselines3.common.policies.ActorCriticCnnPolicy):
|
|
||||||
|
|
||||||
def __init__(self, observation_space, *args, **kwargs):
|
|
||||||
super().__init__(observation_space['obs'], *args, **kwargs)
|
|
||||||
|
|
||||||
def _prior_distribution(self, s):
|
|
||||||
latent_pi, latent_vf, latent_sde = self._get_latent(s)
|
|
||||||
distribution = self._get_action_dist_from_latent(latent_pi, latent_sde)
|
|
||||||
values = self.value_net(latent_vf)
|
|
||||||
return values, distribution.distribution
|
|
||||||
|
|
||||||
def predict(self, obs, eps=1e-6):
|
|
||||||
s, m = obs['obs'], obs['mask']
|
|
||||||
values, prior = self._prior_distribution(s)
|
|
||||||
posterior = Categorical((prior.probs + eps) * m)
|
|
||||||
ch = posterior.sample()
|
|
||||||
return ch, values, posterior.log_prob(ch)
|
|
||||||
|
|
||||||
def evaluate_actions(self, obs, ch, eps=1e-6):
|
|
||||||
s, m = obs['obs'], obs['mask']
|
|
||||||
values, prior = self._prior_distribution(s)
|
|
||||||
posterior = Categorical((prior.probs + eps) * m)
|
|
||||||
return values, posterior.log_prob(ch), posterior.entropy() # additional values used by PPO.train
|
|
||||||
|
|
||||||
class OptionsEnv(gym.Wrapper):
|
class OptionsEnv(gym.Wrapper):
|
||||||
|
"""
|
||||||
|
Wrap an intersimple environment with an options generator
|
||||||
|
"""
|
||||||
def __init__(self, env, *args, **kwargs):
|
def __init__(self, env, *args, **kwargs):
|
||||||
|
"""
|
||||||
|
Initialize wrapped environment and set high-level action and observation spaces
|
||||||
|
"""
|
||||||
super().__init__(env, *args, **kwargs)
|
super().__init__(env, *args, **kwargs)
|
||||||
num_hl_options = len(ALL_OPTIONS)
|
num_hl_options = len(ALL_OPTIONS)
|
||||||
self.action_space = gym.spaces.Discrete(num_hl_options)
|
self.action_space = gym.spaces.Discrete(num_hl_options)
|
||||||
@@ -68,6 +53,13 @@ class OptionsEnv(gym.Wrapper):
|
|||||||
raise NotImplementedError('Use `LLOptions` or `HLOptions` for sampling.')
|
raise NotImplementedError('Use `LLOptions` or `HLOptions` for sampling.')
|
||||||
|
|
||||||
def sample(self, generator):
|
def sample(self, generator):
|
||||||
|
"""
|
||||||
|
yield transitions using a generator
|
||||||
|
Args:
|
||||||
|
generator (sb3.PPO)
|
||||||
|
Yields:
|
||||||
|
|
||||||
|
"""
|
||||||
self.done = True
|
self.done = True
|
||||||
while True:
|
while True:
|
||||||
self.episode_start = False
|
self.episode_start = False
|
||||||
@@ -104,13 +96,23 @@ class LLOptions(OptionsEnv):
|
|||||||
"""Sample low-level (state, action) tuples for discriminator training."""
|
"""Sample low-level (state, action) tuples for discriminator training."""
|
||||||
|
|
||||||
def __init__(self, *args, **kwargs):
|
def __init__(self, *args, **kwargs):
|
||||||
|
"""
|
||||||
|
LLOption uses the true LL observations
|
||||||
|
"""
|
||||||
super().__init__(*args, **kwargs)
|
super().__init__(*args, **kwargs)
|
||||||
|
# overwrite observation space to just output obs directly
|
||||||
self.observation_space = self.observation_space['obs']
|
self.observation_space = self.observation_space['obs']
|
||||||
|
|
||||||
def _after_choice(self):
|
def _after_choice(self):
|
||||||
|
"""
|
||||||
|
After each option choice, initialize/reset the transition buffer
|
||||||
|
"""
|
||||||
self._transition_buffer = []
|
self._transition_buffer = []
|
||||||
|
|
||||||
def _after_step(self):
|
def _after_step(self):
|
||||||
|
"""
|
||||||
|
After each ll action, append s, s', a, done to transition buffer
|
||||||
|
"""
|
||||||
self._transition_buffer.append({
|
self._transition_buffer.append({
|
||||||
'obs': self.s,
|
'obs': self.s,
|
||||||
'next_obs': self.nexts,
|
'next_obs': self.nexts,
|
||||||
@@ -119,9 +121,18 @@ class LLOptions(OptionsEnv):
|
|||||||
})
|
})
|
||||||
|
|
||||||
def _transitions(self):
|
def _transitions(self):
|
||||||
|
"""
|
||||||
|
Yield from the transition buffer
|
||||||
|
"""
|
||||||
yield from self._transition_buffer
|
yield from self._transition_buffer
|
||||||
|
|
||||||
def sample_ll(self, policy):
|
def sample_ll(self, policy):
|
||||||
|
"""
|
||||||
|
Args:
|
||||||
|
policy
|
||||||
|
Returns:
|
||||||
|
gen: iterable which samples low-level transitions from the environment
|
||||||
|
"""
|
||||||
return self.sample(policy)
|
return self.sample(policy)
|
||||||
|
|
||||||
class HLOptions(OptionsEnv):
|
class HLOptions(OptionsEnv):
|
||||||
@@ -131,11 +142,17 @@ class HLOptions(OptionsEnv):
|
|||||||
super().__init__(*args, **kwargs)
|
super().__init__(*args, **kwargs)
|
||||||
|
|
||||||
def _after_choice(self):
|
def _after_choice(self):
|
||||||
|
"""
|
||||||
|
After an option selection, initialize total reward and number of steps
|
||||||
|
"""
|
||||||
self.obs = {'obs': np.copy(self.s), 'mask': np.copy(self.m)}
|
self.obs = {'obs': np.copy(self.s), 'mask': np.copy(self.m)}
|
||||||
self.r = 0
|
self.r = 0
|
||||||
self.steps = 0
|
self.steps = 0
|
||||||
|
|
||||||
def _after_step(self):
|
def _after_step(self):
|
||||||
|
"""
|
||||||
|
After each low-level action, add the discounted discriminated reward score (given a discriminator)
|
||||||
|
"""
|
||||||
self.r += self.discount**self.steps * self.discriminator.discrim_net.reward_train(
|
self.r += self.discount**self.steps * self.discriminator.discrim_net.reward_train(
|
||||||
state=torch.tensor(self.s).unsqueeze(0).to(self.discriminator.discrim_net.device()),
|
state=torch.tensor(self.s).unsqueeze(0).to(self.discriminator.discrim_net.device()),
|
||||||
action=torch.tensor([[self.a]]).to(self.discriminator.discrim_net.device()),
|
action=torch.tensor([[self.a]]).to(self.discriminator.discrim_net.device()),
|
||||||
@@ -145,6 +162,18 @@ class HLOptions(OptionsEnv):
|
|||||||
self.steps += 1
|
self.steps += 1
|
||||||
|
|
||||||
def _transitions(self):
|
def _transitions(self):
|
||||||
|
"""
|
||||||
|
Yield a single dictionary per high-level selected action
|
||||||
|
Fields:
|
||||||
|
obs: high-level state and mask at selection
|
||||||
|
action: chosen high-level action
|
||||||
|
reward: accumulated option reward
|
||||||
|
episode_start: whether the action was chosen at the episode start
|
||||||
|
value: the value estimate from the starting state
|
||||||
|
log_prob: the log_prob of the selected action from the starting state
|
||||||
|
done: whether the episode has ended
|
||||||
|
|
||||||
|
"""
|
||||||
yield {
|
yield {
|
||||||
'obs': self.obs,
|
'obs': self.obs,
|
||||||
'action': self.ch,
|
'action': self.ch,
|
||||||
@@ -156,16 +185,29 @@ class HLOptions(OptionsEnv):
|
|||||||
}
|
}
|
||||||
|
|
||||||
def sample_hl(self, policy, discriminator):
|
def sample_hl(self, policy, discriminator):
|
||||||
|
"""
|
||||||
|
Args:
|
||||||
|
policy
|
||||||
|
discriminator: function with which to score rewards
|
||||||
|
Returns:
|
||||||
|
gen: iterable which samples high-level transitions from the environment
|
||||||
|
"""
|
||||||
self.discriminator = discriminator
|
self.discriminator = discriminator
|
||||||
return self.sample(policy)
|
return self.sample(policy)
|
||||||
|
|
||||||
class RenderOptions(LLOptions):
|
class RenderOptions(LLOptions):
|
||||||
|
|
||||||
def _after_step(self):
|
def _after_step(self):
|
||||||
|
"""
|
||||||
|
Render the environment after each low-level step
|
||||||
|
"""
|
||||||
super()._after_step()
|
super()._after_step()
|
||||||
self.env.render()
|
self.env.render()
|
||||||
|
|
||||||
def close(self, *args, **kwargs):
|
def close(self, *args, **kwargs):
|
||||||
|
"""
|
||||||
|
On 'close', close the environment
|
||||||
|
"""
|
||||||
self.env.close(*args, **kwargs)
|
self.env.close(*args, **kwargs)
|
||||||
|
|
||||||
def available_actions(env):
|
def available_actions(env):
|
||||||
@@ -195,103 +237,6 @@ def generate_plan(env, i):
|
|||||||
assert len(plan) == t, "incorrect plan length"
|
assert len(plan) == t, "incorrect plan length"
|
||||||
return plan
|
return plan
|
||||||
|
|
||||||
def check_future_collisions_circle(env, actions):
|
|
||||||
"""Compute collision information for circular vehicle approximations
|
|
||||||
|
|
||||||
Args:
|
|
||||||
env (gym.Env): current environment state
|
|
||||||
actions (list of torch.Tensor): list of B (T, nv, adims) T-length action profiles
|
|
||||||
Returns:
|
|
||||||
states (torch.Tensor): tensor of shape (B, T, nv, 5) of future states based on the action profiles
|
|
||||||
collision_tensor (torch.Tensor): tensor of shape (B, T, nv) of bools indicating which plan collides with which vehicles in which time frame
|
|
||||||
false: colliding, true: not colliding
|
|
||||||
"""
|
|
||||||
B, (T, nv, _) = len(actions), actions[0].shape
|
|
||||||
|
|
||||||
states = torch.stack(env._env.propagate_action_profile(actions), axis=0)
|
|
||||||
assert states.shape == (B, T, nv, 5)
|
|
||||||
|
|
||||||
distance = ((states[:, :, :, :2] - states[:, :, env._agent:env._agent+1, :2])**2).sum(-1).sqrt()
|
|
||||||
distance = torch.where(distance.isnan(), np.inf*torch.ones_like(distance), distance) # only collide with spawned agents
|
|
||||||
distance[:, :, env._agent] = np.inf # cannot collide with itself
|
|
||||||
assert distance.shape == (B, T, nv)
|
|
||||||
|
|
||||||
radius = (env._env._lengths**2 + env._env._widths**2).sqrt() / 2
|
|
||||||
min_distance = radius[env._agent] + radius
|
|
||||||
min_distance = min_distance.unsqueeze(0).unsqueeze(0)
|
|
||||||
assert min_distance.shape == (1, 1, nv)
|
|
||||||
|
|
||||||
collision_tensor = distance > min_distance
|
|
||||||
assert collision_tensor.shape == (B, T, nv)
|
|
||||||
return states, collision_tensor
|
|
||||||
|
|
||||||
def check_future_collisions_fast(env, actions):
|
|
||||||
"""Checks whether `env._agent` would collide with other agents assuming `actions` as input.
|
|
||||||
|
|
||||||
Vehicles are (over-)approximated by single circles.
|
|
||||||
|
|
||||||
Args:
|
|
||||||
env (gym.Env): current environment state
|
|
||||||
actions (list of torch.Tensor): list of B (T, nv, adims) T-length action profiles
|
|
||||||
Returns:
|
|
||||||
feasible (torch.Tensor): tensor of shape (B,) indicating whether the respective action profiles are collision-free
|
|
||||||
"""
|
|
||||||
_, collision_tensor = check_future_collisions_circle(env, actions)
|
|
||||||
return collision_tensor.all(-1).all(-1)
|
|
||||||
|
|
||||||
def check_future_collisions_exact(env, actions):
|
|
||||||
"""
|
|
||||||
Checks whether `env._agent` would collide with other agents assuming `actions` as input.
|
|
||||||
|
|
||||||
Args:
|
|
||||||
env (gym.Env): current environment state
|
|
||||||
actions (list of torch.Tensor): list of B (T, nv, adims) T-length action profiles
|
|
||||||
Returns:
|
|
||||||
feasible (torch.Tensor): tensor of shape (B,) indicating whether the respective action profiles are collision-free
|
|
||||||
"""
|
|
||||||
# First check with simple circle collision check
|
|
||||||
states, collision_tensor = check_future_collisions_circle(env, actions)
|
|
||||||
(B, T, nv, _) = states.shape
|
|
||||||
# For those that have colliding circles, check exactly
|
|
||||||
colliding_mask = ~collision_tensor
|
|
||||||
|
|
||||||
ego_states = states[:, :, env._agent:env._agent+1, :].expand(states.shape)
|
|
||||||
assert ego_states.shape == states.shape
|
|
||||||
|
|
||||||
# get dimensions
|
|
||||||
lengths = env._env._lengths.expand(states.shape[:3])
|
|
||||||
widths = env._env._widths.expand(states.shape[:3])
|
|
||||||
ego_lengths = lengths[:, :, env._agent:env._agent+1].expand(lengths.shape)
|
|
||||||
ego_widths = widths[:, :, env._agent:env._agent+1].expand(widths.shape)
|
|
||||||
assert lengths.shape == widths.shape == ego_lengths.shape == ego_widths.shape == (B, T, nv)
|
|
||||||
|
|
||||||
# For every collision instance between ego and other vehicle, check whether rectangles intersect
|
|
||||||
exact_collisions = torch.zeros_like(collision_tensor[colliding_mask])
|
|
||||||
for i, (ego_state, ego_length, ego_width, other_state, other_length, other_width) in enumerate(zip(
|
|
||||||
ego_states[colliding_mask], ego_lengths[colliding_mask], ego_widths[colliding_mask],
|
|
||||||
states[colliding_mask], lengths[colliding_mask], widths[colliding_mask]
|
|
||||||
)):
|
|
||||||
assert ego_state.shape == other_state.shape == (5,)
|
|
||||||
assert ego_length.shape == ego_width.shape == other_length.shape == other_width.shape == ()
|
|
||||||
p_ego = state_to_polygon(ego_state, ego_length, ego_width)
|
|
||||||
p_other = state_to_polygon(other_state, other_length, other_width)
|
|
||||||
exact_collisions[i] = p_ego.intersects(p_other)
|
|
||||||
|
|
||||||
collision_tensor[colliding_mask] = ~exact_collisions
|
|
||||||
return collision_tensor.all(-1).all(-1)
|
|
||||||
|
|
||||||
def feasible(env, plan, ch):
|
|
||||||
"""Check if input profile is feasible given current `env` state. Action `ch=0` is safe fallback."""
|
|
||||||
if ch == 0:
|
|
||||||
return True
|
|
||||||
|
|
||||||
# zero pad plan - Take (T,) np plan and convert it to (T, nv, 1) torch.Tensor
|
|
||||||
full_plan = torch.zeros(len(plan), env._env._nv, 1)
|
|
||||||
full_plan[:, env._agent, 0] = torch.tensor(plan)
|
|
||||||
# valid = check_future_collisions_fast(env, [full_plan]) # check_future_collisions_fast takes in B-list and outputs (B,) bool tensor
|
|
||||||
valid = check_future_collisions_exact(env, [full_plan]) # check_future_collisions_fast takes in B-list and outputs (B,) bool tensor
|
|
||||||
return valid.item()
|
|
||||||
|
|
||||||
def flatten_transitions(transitions):
|
def flatten_transitions(transitions):
|
||||||
return {
|
return {
|
||||||
'obs': np.stack(list(t['obs'] for t in transitions), axis=0),
|
'obs': np.stack(list(t['obs'] for t in transitions), axis=0),
|
||||||
@@ -326,8 +271,20 @@ def train_generator(env, generator, discriminator, num_samples):
|
|||||||
|
|
||||||
generator.train()
|
generator.train()
|
||||||
|
|
||||||
def train(expert_data, epochs=20, expert_batch_size=32, generator_steps=1024, discount=0.99):
|
def train(expert_data, env_class=NRasterizedRandomAgent, env_settings={}, epochs=10, discrim_batch_size=32, generator_steps=2048, discount=0.99):
|
||||||
env = NRasterized(**env_settings)
|
"""
|
||||||
|
Args:
|
||||||
|
expert_data: list of transitions
|
||||||
|
env_class: environment class
|
||||||
|
env_settings: environment settings
|
||||||
|
epochs: number of epochs to train for
|
||||||
|
discrim_batch_size: discriminator batch size
|
||||||
|
generator_steps: number of steps taken in generator
|
||||||
|
discount: discount factor
|
||||||
|
Returns:
|
||||||
|
generator (stable_baselines3.PPO): options policy
|
||||||
|
"""
|
||||||
|
env = env_class(**env_settings)
|
||||||
env.discount = discount
|
env.discount = discount
|
||||||
|
|
||||||
tempdir = tempfile.TemporaryDirectory(prefix="quickstart")
|
tempdir = tempfile.TemporaryDirectory(prefix="quickstart")
|
||||||
@@ -335,10 +292,10 @@ def train(expert_data, epochs=20, expert_batch_size=32, generator_steps=1024, di
|
|||||||
logger.configure(tempdir_path / "GAIL/")
|
logger.configure(tempdir_path / "GAIL/")
|
||||||
print(f"All Tensorboards and logging are being written inside {tempdir_path}/.")
|
print(f"All Tensorboards and logging are being written inside {tempdir_path}/.")
|
||||||
|
|
||||||
venv = make_vec_env(NRasterized, n_envs=1, env_kwargs=env_settings)
|
venv = make_vec_env(env_class, n_envs=1, env_kwargs=env_settings)
|
||||||
discriminator = adversarial.GAIL(
|
discriminator = adversarial.GAIL(
|
||||||
expert_data=expert_data,
|
expert_data=expert_data,
|
||||||
expert_batch_size=expert_batch_size,
|
expert_batch_size=discrim_batch_size,
|
||||||
discrim_kwargs={'discrim_net': CnnDiscriminatorFlatAction(venv)},
|
discrim_kwargs={'discrim_net': CnnDiscriminatorFlatAction(venv)},
|
||||||
#discrim_kwargs={'discrim_net': CnnDiscriminator(venv)},
|
#discrim_kwargs={'discrim_net': CnnDiscriminator(venv)},
|
||||||
venv=venv, # unused
|
venv=venv, # unused
|
||||||
@@ -360,7 +317,7 @@ def train(expert_data, epochs=20, expert_batch_size=32, generator_steps=1024, di
|
|||||||
)
|
)
|
||||||
|
|
||||||
for _ in tqdm(range(epochs)):
|
for _ in tqdm(range(epochs)):
|
||||||
train_discriminator(LLOptions(env), generator, discriminator, num_samples=expert_batch_size)
|
train_discriminator(LLOptions(env), generator, discriminator, num_samples=discrim_batch_size)
|
||||||
train_generator(HLOptions(env), generator, discriminator, num_samples=generator_steps)
|
train_generator(HLOptions(env), generator, discriminator, num_samples=generator_steps)
|
||||||
|
|
||||||
return generator
|
return generator
|
||||||
@@ -368,18 +325,31 @@ def train(expert_data, epochs=20, expert_batch_size=32, generator_steps=1024, di
|
|||||||
# %%
|
# %%
|
||||||
if __name__ == '__main__':
|
if __name__ == '__main__':
|
||||||
# %%
|
# %%
|
||||||
|
model_name = 'gail_options_image'
|
||||||
with open("data/NormalizedIntersimpleExpertMu.001_NRasterizedAgent51w36h36mppx2.pkl", "rb") as f:
|
env_class = NRasterizedRandomAgent
|
||||||
trajectories = pickle.load(f)
|
env_settings = {'width': 36, 'height': 36, 'm_per_px': 2}
|
||||||
transitions = rollout.flatten_trajectories(trajectories)
|
|
||||||
generator = train(transitions)
|
#env_class = NRasterized
|
||||||
|
#env_settings = {'agent': 51, 'width': 36, 'height': 36, 'm_per_px': 2}
|
||||||
|
files = ['../../../expert_data/DR_USA_Roundabout_FT0/track%04i/expert.pkl'%(i) for i in range(5)]
|
||||||
|
transitions=load_experts(files)
|
||||||
|
|
||||||
|
generator = train(
|
||||||
|
transitions,
|
||||||
|
env_class=env_class,
|
||||||
|
env_settings=env_settings,
|
||||||
|
epochs=10,
|
||||||
|
discrim_batch_size=32,
|
||||||
|
generator_steps=2048,
|
||||||
|
discount=0.99
|
||||||
|
)
|
||||||
|
|
||||||
generator.save(model_name)
|
generator.save(model_name)
|
||||||
|
|
||||||
# %%
|
# %%
|
||||||
model = stable_baselines3.PPO.load(model_name)
|
model = stable_baselines3.PPO.load(model_name)
|
||||||
|
|
||||||
env = RenderOptions(NRasterized(**env_settings))
|
env = RenderOptions(NRasterizedRandomAgent(**env_args))
|
||||||
|
|
||||||
for s in env.sample_ll(model):
|
for s in env.sample_ll(model):
|
||||||
if s['dones']:
|
if s['dones']:
|
||||||
|
|||||||
@@ -1,3 +1,3 @@
|
|||||||
from src.expert_data import generate_expert_data, load_expert_data
|
from src.data.expert_data import generate_expert_data, load_expert_data
|
||||||
from src.data_utils import InteractionDatasetSingleAgent
|
from src.data.data_utils import InteractionDatasetSingleAgent
|
||||||
from src.metrics import metrics
|
from src.metrics import metrics
|
||||||
1
src/data/__init__.py
Normal file
1
src/data/__init__.py
Normal file
@@ -0,0 +1 @@
|
|||||||
|
from src.data.expert import demonstrations, load_experts, process_experts
|
||||||
@@ -1,7 +1,7 @@
|
|||||||
import torch
|
import torch
|
||||||
from torch.utils.data import Dataset
|
from torch.utils.data import Dataset
|
||||||
import numpy as np
|
import numpy as np
|
||||||
from src.expert_data import load_expert_data
|
from src.data.expert_data import load_expert_data
|
||||||
import os
|
import os
|
||||||
opj = os.path.join
|
opj = os.path.join
|
||||||
|
|
||||||
215
src/data/expert.py
Normal file
215
src/data/expert.py
Normal file
@@ -0,0 +1,215 @@
|
|||||||
|
from intersim.envs.intersimple import Intersimple
|
||||||
|
from stable_baselines3.common.policies import BasePolicy
|
||||||
|
import gym
|
||||||
|
import intersim.envs.intersimple
|
||||||
|
import pickle
|
||||||
|
from tqdm import tqdm
|
||||||
|
import imitation.data.rollout as rollout
|
||||||
|
from stable_baselines3.common.vec_env.dummy_vec_env import DummyVecEnv
|
||||||
|
from imitation.data.wrappers import RolloutInfoWrapper
|
||||||
|
import copy
|
||||||
|
import os
|
||||||
|
|
||||||
|
class IntersimExpert(BasePolicy):
|
||||||
|
|
||||||
|
def __init__(self, intersim_env, mu=0, *args, **kwargs):
|
||||||
|
super().__init__(
|
||||||
|
observation_space=gym.spaces.Space(),
|
||||||
|
action_space=gym.spaces.Space(),
|
||||||
|
*args, **kwargs
|
||||||
|
)
|
||||||
|
self._intersim = intersim_env
|
||||||
|
self._mu = mu
|
||||||
|
|
||||||
|
def forward(self, *args, **kwargs):
|
||||||
|
raise NotImplementedError()
|
||||||
|
|
||||||
|
def _predict(self, *args, **kwargs):
|
||||||
|
raise NotImplementedError()
|
||||||
|
|
||||||
|
def _action(self):
|
||||||
|
target_t = min(self._intersim._ind + 1, len(self._intersim._svt.simstate) - 1)
|
||||||
|
target_state = self._intersim._svt.simstate[target_t]
|
||||||
|
return self._intersim.target_state(target_state, mu=self._mu)
|
||||||
|
|
||||||
|
def predict(self, *args, **kwargs):
|
||||||
|
return self._action(), None
|
||||||
|
|
||||||
|
class IntersimpleExpert(BasePolicy):
|
||||||
|
|
||||||
|
def __init__(self, intersimple_env, mu=0, *args, **kwargs):
|
||||||
|
super().__init__(
|
||||||
|
observation_space=intersimple_env.observation_space,
|
||||||
|
action_space=intersimple_env.action_space,
|
||||||
|
*args, **kwargs
|
||||||
|
)
|
||||||
|
self._intersimple = intersimple_env
|
||||||
|
self._intersim_expert = IntersimExpert(intersimple_env._env, mu=mu)
|
||||||
|
|
||||||
|
def forward(self, *args, **kwargs):
|
||||||
|
raise NotImplementedError()
|
||||||
|
|
||||||
|
def _predict(self, *args, **kwargs):
|
||||||
|
raise NotImplementedError()
|
||||||
|
|
||||||
|
def _action(self):
|
||||||
|
return self._intersim_expert._action()[self._intersimple._agent]
|
||||||
|
|
||||||
|
def predict(self, *args, **kwargs):
|
||||||
|
return self._action(), None
|
||||||
|
|
||||||
|
class NormalizedIntersimpleExpert(IntersimpleExpert):
|
||||||
|
|
||||||
|
def __init__(self, *args, **kwargs):
|
||||||
|
super().__init__(*args, **kwargs)
|
||||||
|
|
||||||
|
def predict(self, *args, **kwargs):
|
||||||
|
action, _ = super().predict(*args, **kwargs)
|
||||||
|
return self._intersimple._normalize(action), None
|
||||||
|
|
||||||
|
class DummyVecEnvPolicy(BasePolicy):
|
||||||
|
|
||||||
|
def __init__(self, experts):
|
||||||
|
self._experts = [e() for e in experts]
|
||||||
|
|
||||||
|
def forward(self, *args, **kwargs):
|
||||||
|
raise NotImplementedError()
|
||||||
|
|
||||||
|
def _predict(self, *args, **kwargs):
|
||||||
|
raise NotImplementedError()
|
||||||
|
|
||||||
|
def predict(self, *args, **kwargs):
|
||||||
|
predictions = [e.predict() for e in self._experts]
|
||||||
|
actions = [p[0] for p in predictions]
|
||||||
|
states = [p[1] for p in predictions]
|
||||||
|
return actions, states
|
||||||
|
|
||||||
|
def forward(self, *args, **kwargs):
|
||||||
|
raise NotImplementedError()
|
||||||
|
|
||||||
|
def _predict(self, *args, **kwargs):
|
||||||
|
raise NotImplementedError()
|
||||||
|
|
||||||
|
def save_video(env, expert):
|
||||||
|
env.reset()
|
||||||
|
env.render()
|
||||||
|
done = False
|
||||||
|
while not done:
|
||||||
|
actions, _ = expert.predict()
|
||||||
|
_, _, done, _ = env.step(actions)
|
||||||
|
env.render()
|
||||||
|
env.close()
|
||||||
|
|
||||||
|
def load_experts(expert_files):
|
||||||
|
"""
|
||||||
|
Load expert trajectories from files and combine their transitions into a single RB
|
||||||
|
|
||||||
|
Args:
|
||||||
|
expert_files (list): list of expert file strings
|
||||||
|
Returns:
|
||||||
|
transitions (list): list of combined expert episode transitions
|
||||||
|
"""
|
||||||
|
trajectories = []
|
||||||
|
for file in tqdm(expert_files):
|
||||||
|
with open(file, "rb") as f:
|
||||||
|
new_trajectories = pickle.load(f)
|
||||||
|
trajectories += new_trajectories
|
||||||
|
transitions = rollout.flatten_trajectories(trajectories)
|
||||||
|
return transitions
|
||||||
|
|
||||||
|
def demonstrations(expert='NormalizedIntersimpleExpert', env='NRasterizedIncrementingAgent', path=None, min_timesteps=None, min_episodes=None, video=False, env_args={}, policy_args={}):
|
||||||
|
"""Rollout and save expert demos.
|
||||||
|
|
||||||
|
Usage:
|
||||||
|
python -m intersimple.expert <flags>
|
||||||
|
Args:
|
||||||
|
expert (class): class of expert
|
||||||
|
env (class): class of env intersim.envs.intersimple
|
||||||
|
path (str): path to store output
|
||||||
|
min_timesteps (int): min number of timesteps for call to rollout.rollout_and_save
|
||||||
|
min_episodes (int): min number of episodes for call to rollout.rollout_and_save
|
||||||
|
video (bool): whether to save a video of the expert until a single environment instantiation stops
|
||||||
|
env_args (dict): dictionary of kwargs when instantiating environment class
|
||||||
|
policy_args (dict): dictionary of kwargs when instantiating Expert policy
|
||||||
|
"""
|
||||||
|
|
||||||
|
Env = intersim.envs.intersimple.__dict__[env]
|
||||||
|
Expert = globals()[expert]
|
||||||
|
|
||||||
|
env = Env(**env_args)
|
||||||
|
info_env = RolloutInfoWrapper(env) # getting rollout info (dictionary) from environment
|
||||||
|
venv = DummyVecEnv([lambda: info_env]) # making a DummyVecEnv with a list of a function that when called returns the rollout info
|
||||||
|
|
||||||
|
policy = Expert(env, **policy_args) # instantiate an expert policy from specified class with instantiated environment and policy kwargs
|
||||||
|
venv_policy = DummyVecEnvPolicy([lambda: policy]) # make a DummyVecEnvPolicy with a list of a function that when called returns the Expert policy
|
||||||
|
|
||||||
|
if min_timesteps is None and min_episodes is None:
|
||||||
|
min_episodes = env.nv # one episode per vehicle being controlled in environment (hopefully an incrementing agent environment)
|
||||||
|
|
||||||
|
if video:
|
||||||
|
save_video(env, policy)
|
||||||
|
|
||||||
|
path = path or (policy.__class__.__name__ + '_' + env.__class__.__name__ + '.pkl')
|
||||||
|
suntil = rollout.make_sample_until(
|
||||||
|
min_timesteps=min_timesteps,
|
||||||
|
min_episodes=min_episodes,
|
||||||
|
)
|
||||||
|
rollout.rollout_and_save(
|
||||||
|
path=path,
|
||||||
|
policy=venv_policy,
|
||||||
|
venv=venv,
|
||||||
|
sample_until=suntil
|
||||||
|
)
|
||||||
|
|
||||||
|
def process_experts(filename:str='expert.pkl',
|
||||||
|
locs:list=None,
|
||||||
|
tracks:list=None,
|
||||||
|
env_class:str='NRasterizedIncrementingAgent',
|
||||||
|
env_args:dict={'width':36,'height':36,'m_per_px':2},
|
||||||
|
expert_class:str='NormalizedIntersimpleExpert',
|
||||||
|
expert_args:dict={'mu':0.001}):
|
||||||
|
"""
|
||||||
|
Process all experts in the Interaction Dataset
|
||||||
|
For now, using NormalizedIntersimpleExpert with NRasterizedIncrementingAgent environment
|
||||||
|
|
||||||
|
Args:
|
||||||
|
filename (str): name for track file
|
||||||
|
locs (list): list of location ids
|
||||||
|
tracks (list): list of track numbers
|
||||||
|
env_class (str): class of environment
|
||||||
|
env_args (dict): default environment kwargs
|
||||||
|
expert_class (str): class of expert
|
||||||
|
expert_args (dict): default expert kwargs
|
||||||
|
"""
|
||||||
|
locs = locs or intersim.LOCATIONS
|
||||||
|
tracks = tracks or range(intersim.MAX_TRACKS)
|
||||||
|
pbar = tqdm(total=len(locs)*len(tracks))
|
||||||
|
for loc in locs:
|
||||||
|
for track in tracks:
|
||||||
|
|
||||||
|
iloc = intersim.LOCATIONS.index(loc)
|
||||||
|
|
||||||
|
it_env_args = copy.deepcopy(env_args)
|
||||||
|
it_env_args.update({
|
||||||
|
'loc':iloc,
|
||||||
|
'track':track,
|
||||||
|
})
|
||||||
|
out_folder = os.path.join('expert_data',loc, 'track%04i'%(track))
|
||||||
|
if not os.path.isdir(out_folder):
|
||||||
|
os.makedirs(out_folder)
|
||||||
|
it_path = os.path.join(out_folder,filename)
|
||||||
|
|
||||||
|
demonstrations(
|
||||||
|
expert=expert_class,
|
||||||
|
env=env_class,
|
||||||
|
path=it_path,
|
||||||
|
env_args=it_env_args,
|
||||||
|
policy_args=expert_args,
|
||||||
|
)
|
||||||
|
pbar.update(1)
|
||||||
|
pbar.close()
|
||||||
|
|
||||||
|
if __name__=='__main__':
|
||||||
|
import fire
|
||||||
|
fire.Fire(process_experts)
|
||||||
|
|
||||||
1
src/discriminator/__init__.py
Normal file
1
src/discriminator/__init__.py
Normal file
@@ -0,0 +1 @@
|
|||||||
|
from src.discriminator.discriminator import *
|
||||||
101
src/discriminator/discriminator.py
Normal file
101
src/discriminator/discriminator.py
Normal file
@@ -0,0 +1,101 @@
|
|||||||
|
import torch
|
||||||
|
|
||||||
|
# imitation.rewards.discrim_nets.DiscrimNetGAIL is composed of self.discriminator (nn.Module),
|
||||||
|
# which gets called with inputs (state, action) when needed.
|
||||||
|
|
||||||
|
class CnnDiscriminator(torch.nn.Module):
|
||||||
|
"""ConvNet similar to stable_baselines3.common.policies.ActorCriticCnnPolicy."""
|
||||||
|
|
||||||
|
def __init__(self, env):
|
||||||
|
super().__init__()
|
||||||
|
|
||||||
|
obs_channels, _, _ = env.observation_space.shape
|
||||||
|
(action_size,) = env.action_space.shape
|
||||||
|
in_channels = obs_channels + action_size
|
||||||
|
|
||||||
|
self.cnn = torch.nn.Sequential(
|
||||||
|
torch.nn.Conv2d(in_channels, 32, kernel_size=(8, 8), stride=(4, 4)), # 5+1 -> 32
|
||||||
|
torch.nn.ReLU(),
|
||||||
|
torch.nn.Conv2d(32, 64, kernel_size=(4, 4), stride=(2, 2)), # 32 -> 64
|
||||||
|
torch.nn.ReLU(),
|
||||||
|
torch.nn.Conv2d(64, 64, kernel_size=(3, 3), stride=(1, 1)), # 64 -> 64
|
||||||
|
torch.nn.ReLU(),
|
||||||
|
torch.nn.Flatten(start_dim=1, end_dim=-1),
|
||||||
|
torch.nn.LazyLinear(512), # 28224 -> 512
|
||||||
|
torch.nn.ReLU(),
|
||||||
|
torch.nn.LazyLinear(1), # 512 -> 1
|
||||||
|
)
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _concatenate(state, action):
|
||||||
|
b, _, h, w = state.shape
|
||||||
|
_, a = action.shape
|
||||||
|
act = action.unsqueeze(-1).unsqueeze(-1).expand((b, a, h, w))
|
||||||
|
sa = torch.cat((state, act), -3)
|
||||||
|
return sa
|
||||||
|
|
||||||
|
def forward(self, state, action):
|
||||||
|
sa = self._concatenate(state, action)
|
||||||
|
assert sa.ndim == 4
|
||||||
|
return self.cnn(sa).squeeze(1)
|
||||||
|
|
||||||
|
class CnnDiscriminatorFlatAction(torch.nn.Module):
|
||||||
|
"""ConvNet similar to stable_baselines3.common.policies.ActorCriticCnnPolicy."""
|
||||||
|
|
||||||
|
def __init__(self, env):
|
||||||
|
super().__init__()
|
||||||
|
|
||||||
|
obs_channels, _, _ = env.observation_space.shape
|
||||||
|
(action_size,) = env.action_space.shape
|
||||||
|
in_channels = obs_channels
|
||||||
|
|
||||||
|
self.cnn = torch.nn.Sequential(
|
||||||
|
torch.nn.Conv2d(in_channels, 32, kernel_size=(8, 8), stride=(4, 4)), # in_channels -> 32
|
||||||
|
torch.nn.ReLU(),
|
||||||
|
torch.nn.Conv2d(32, 64, kernel_size=(4, 4), stride=(2, 2)), # 32 -> 64
|
||||||
|
torch.nn.ReLU(),
|
||||||
|
torch.nn.Conv2d(64, 64, kernel_size=(3, 3), stride=(1, 1)), # 64 -> 64
|
||||||
|
torch.nn.ReLU(),
|
||||||
|
torch.nn.Flatten(start_dim=1, end_dim=-1),
|
||||||
|
torch.nn.LazyLinear(128), # 28224 -> 128
|
||||||
|
)
|
||||||
|
self.decoder = torch.nn.Sequential(
|
||||||
|
torch.nn.LazyLinear(64), #128 + 2 -> 64
|
||||||
|
torch.nn.ReLU(),
|
||||||
|
torch.nn.LazyLinear(64), #64 -> 64
|
||||||
|
torch.nn.ReLU(),
|
||||||
|
torch.nn.LazyLinear(1) #64 -> 1
|
||||||
|
)
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _concatenate(state, action):
|
||||||
|
b, s= state.shape
|
||||||
|
b, a = action.shape
|
||||||
|
sa = torch.cat((state, action), -1)
|
||||||
|
return sa
|
||||||
|
|
||||||
|
def forward(self, state, action):
|
||||||
|
s = self.cnn(state.float())
|
||||||
|
sa = self._concatenate(s, action)
|
||||||
|
assert sa.ndim == 2
|
||||||
|
return self.decoder(sa).squeeze(1)
|
||||||
|
|
||||||
|
class MlpDiscriminator(torch.nn.Module):
|
||||||
|
"""MLP similar to stable_baselines3.common.policies.ActorCriticPolicy."""
|
||||||
|
|
||||||
|
def __init__(self, env=None):
|
||||||
|
super().__init__()
|
||||||
|
self.flatten = torch.nn.Flatten(start_dim=1, end_dim=-1)
|
||||||
|
self.mlp = torch.nn.Sequential(
|
||||||
|
torch.nn.LazyLinear(64), # 42 -> 64
|
||||||
|
torch.nn.Tanh(),
|
||||||
|
torch.nn.LazyLinear(64), # 64 -> 64
|
||||||
|
torch.nn.Tanh(),
|
||||||
|
torch.nn.LazyLinear(1), # 64 -> 1
|
||||||
|
)
|
||||||
|
|
||||||
|
def forward(self, state, action):
|
||||||
|
flat = self.flatten(state)
|
||||||
|
sa = torch.cat((action, flat), -1)
|
||||||
|
assert sa.ndim == 2
|
||||||
|
return self.mlp(sa).squeeze(1)
|
||||||
@@ -1 +1,2 @@
|
|||||||
from src.policies.policy import IntersimPolicy, IntersimStateNet, IntersimStateActionNet, generate_transforms
|
from src.policies.policy import IntersimPolicy, IntersimStateNet, IntersimStateActionNet, generate_transforms
|
||||||
|
from src.policies.options import OptionsCnnPolicy
|
||||||
59
src/policies/options.py
Normal file
59
src/policies/options.py
Normal file
@@ -0,0 +1,59 @@
|
|||||||
|
import stable_baselines3
|
||||||
|
from torch.distributions import Categorical
|
||||||
|
|
||||||
|
class OptionsCnnPolicy(stable_baselines3.common.policies.ActorCriticCnnPolicy):
|
||||||
|
"""
|
||||||
|
Class for high-level options policy (generator)
|
||||||
|
"""
|
||||||
|
def __init__(self, observation_space, *args, **kwargs):
|
||||||
|
super().__init__(observation_space['obs'], *args, **kwargs)
|
||||||
|
|
||||||
|
def _prior_distribution(self, s):
|
||||||
|
"""
|
||||||
|
Return prior distribution over high-level options (before masking)
|
||||||
|
Args:
|
||||||
|
s (torch.tensor): observation
|
||||||
|
Returns:
|
||||||
|
values (torch.tensor): values from critic
|
||||||
|
dist (torch.distributions): prior distribution over actions
|
||||||
|
"""
|
||||||
|
latent_pi, latent_vf, latent_sde = self._get_latent(s)
|
||||||
|
distribution = self._get_action_dist_from_latent(latent_pi, latent_sde)
|
||||||
|
values = self.value_net(latent_vf)
|
||||||
|
return values, distribution.distribution
|
||||||
|
|
||||||
|
def predict(self, obs, eps=1e-6):
|
||||||
|
"""
|
||||||
|
Will mask invalid states before making action selections
|
||||||
|
Args:
|
||||||
|
obs: dict with keys:
|
||||||
|
obs (torch.tensor): (*,o) true observations
|
||||||
|
mask (torch.tensor): (*,m) mask over valid actions
|
||||||
|
Returns:
|
||||||
|
ch (torch.tensor): (*,a) sampled actions
|
||||||
|
values (torch.tensor): (*,) predicted value at observation
|
||||||
|
log_probs (torch.tensor): (*,) log probabilities of selected actions
|
||||||
|
"""
|
||||||
|
s, m = obs['obs'], obs['mask']
|
||||||
|
values, prior = self._prior_distribution(s)
|
||||||
|
posterior = Categorical((prior.probs + eps) * m)
|
||||||
|
ch = posterior.sample()
|
||||||
|
return ch, values, posterior.log_prob(ch)
|
||||||
|
|
||||||
|
def evaluate_actions(self, obs, ch, eps=1e-6):
|
||||||
|
"""
|
||||||
|
Evaluate particular actions
|
||||||
|
Args:
|
||||||
|
obs: dict with keys:
|
||||||
|
obs (torch.tensor): (*,o) true observations
|
||||||
|
mask (torch.tensor): (*,m) masks over valid actions
|
||||||
|
ch (torch.tensor): (*,a) selected actions
|
||||||
|
Returns:
|
||||||
|
values (torch.tensor): (*,) predicted value at observation
|
||||||
|
log_probs (torch.tensor): (*,) log probabilities of selected actions
|
||||||
|
ent (torch.tensor): (*,) entropy of each distribution over actions
|
||||||
|
"""
|
||||||
|
s, m = obs['obs'], obs['mask']
|
||||||
|
values, prior = self._prior_distribution(s)
|
||||||
|
posterior = Categorical((prior.probs + eps) * m)
|
||||||
|
return values, posterior.log_prob(ch), posterior.entropy() # additional values used by PPO.train
|
||||||
@@ -0,0 +1,2 @@
|
|||||||
|
from src.util.render_env import *
|
||||||
|
from src.util.collisions import feasible
|
||||||
155
src/util/collisions.py
Normal file
155
src/util/collisions.py
Normal file
@@ -0,0 +1,155 @@
|
|||||||
|
import torch
|
||||||
|
import numpy as np
|
||||||
|
from intersim.collisions import state_to_polygon
|
||||||
|
|
||||||
|
def feasible(env, plan, ch, method='exact'):
|
||||||
|
"""Check if input profile is feasible given current `env` state. Action `ch=0` is safe fallback."""
|
||||||
|
if ch == 0:
|
||||||
|
return True
|
||||||
|
|
||||||
|
# zero pad plan - Take (T,) np plan and convert it to (T, nv, 1) torch.Tensor
|
||||||
|
full_plan = torch.zeros(len(plan), env._env._nv, 1)
|
||||||
|
full_plan[:, env._agent, 0] = torch.tensor(plan)
|
||||||
|
|
||||||
|
# check_future_collisions_fast takes in B-list and outputs (B,) bool tensor
|
||||||
|
if method=='circle':
|
||||||
|
valid = check_future_collisions_fast(env, [full_plan])
|
||||||
|
elif method=='ncircles':
|
||||||
|
valid = check_future_collisions_ncircles(env, [fullplan])
|
||||||
|
elif method=='exact':
|
||||||
|
valid = check_future_collisions_exact(env, [full_plan])
|
||||||
|
else:
|
||||||
|
raise NotImplementedError('Invalid collision-checking method')
|
||||||
|
return valid.item()
|
||||||
|
|
||||||
|
def check_future_collisions_ncircles(env, actions, n_circles:int=2):
|
||||||
|
"""Checks whether `env._agent` would collide with other agents assuming `actions` as input.
|
||||||
|
|
||||||
|
Vehicles are (over-)approximated by multiple circles.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
env (gym.Env): current environment state
|
||||||
|
actions (list of torch.Tensor): list of B (T, nv, adims) T-length action profiles
|
||||||
|
Returns:
|
||||||
|
feasible (torch.Tensor): tensor of shape (B,) indicating whether the respective action profiles are collision-free
|
||||||
|
"""
|
||||||
|
assert n_circles >= 2
|
||||||
|
B, (T, nv, _) = len(actions), actions[0].shape
|
||||||
|
|
||||||
|
states = torch.stack(env._env.propagate_action_profile(actions), axis=0)
|
||||||
|
assert states.shape == (B, T, nv, 5)
|
||||||
|
centers = states[:, :, :, :2]
|
||||||
|
psi = states[:, :, :, 3]
|
||||||
|
lon = torch.stack([psi.cos(), psi.sin()],dim=-1) # (B, T, nv, 2)
|
||||||
|
|
||||||
|
# offset between [-env._env.lengths+env._env.widths/2, env._env.lengths/2-env._env.widths/2]
|
||||||
|
back = (-env._env._lengths/2+env._env._widths/2).unsqueeze(-1) # (nv, 1)
|
||||||
|
length = (env._env._lengths-env._env._widths).unsqueeze(-1) # (nv, 1)
|
||||||
|
diff_d = back + length*(torch.arange(n_circles)/(n_circles-1)).unsqueeze(0) # (nv, n_circles)
|
||||||
|
assert diff_d.shape == (nv, n_circles)
|
||||||
|
|
||||||
|
offsets = diff_d[None, None, :, :, None] * lon[:, :, :, None, :]
|
||||||
|
assert offsets.shape == (B, T, nv, n_circles, 2)
|
||||||
|
|
||||||
|
expanded_centers=centers.unsqueeze(-2) + offsets #(B, T, nv, n_circles, 2)
|
||||||
|
assert expanded_centers.shape == (B, T, nv, n_circles, 2)
|
||||||
|
agent_centers = expanded_centers[:,:,env._agent:env._agent+1,:,:] #(B, T, 1, n_circles, 2)
|
||||||
|
ds = expanded_centers.reshape((B, T, nv*n_circles, 1, 2)) - agent_centers #(B, T, nv*nc,1, 2) - (B, T, 1, nc, 2) = (B, T, nv*nc, nc, 2)
|
||||||
|
|
||||||
|
distance = (ds**2).sum(-1).sqrt().reshape((B, T, nv, n_circles, n_circles)) # (B, T, nv, nc, nc)
|
||||||
|
distance = torch.where(distance.isnan(), np.inf*torch.ones_like(distance), distance) # only collide with spawned agents
|
||||||
|
distance[:, :, env._agent] = np.inf # cannot collide with itself
|
||||||
|
assert distance.shape == (B, T, nv, n_circles, n_circles)
|
||||||
|
|
||||||
|
radius = env._env._widths*np.sqrt(2) / 2
|
||||||
|
min_distance = radius[env._agent] + radius
|
||||||
|
min_distance = min_distance[None, None, :, None, None]
|
||||||
|
assert min_distance.shape == (1, 1, nv, 1, 1)
|
||||||
|
|
||||||
|
return (distance > min_distance).all(-1).all(-1).all(-1).all(-1)
|
||||||
|
|
||||||
|
def check_future_collisions_circle(env, actions):
|
||||||
|
"""Compute collision information for circular vehicle approximations
|
||||||
|
|
||||||
|
Args:
|
||||||
|
env (gym.Env): current environment state
|
||||||
|
actions (list of torch.Tensor): list of B (T, nv, adims) T-length action profiles
|
||||||
|
Returns:
|
||||||
|
states (torch.Tensor): tensor of shape (B, T, nv, 5) of future states based on the action profiles
|
||||||
|
collision_tensor (torch.Tensor): tensor of shape (B, T, nv) of bools indicating which plan collides with which vehicles in which time frame
|
||||||
|
false: colliding, true: not colliding
|
||||||
|
"""
|
||||||
|
B, (T, nv, _) = len(actions), actions[0].shape
|
||||||
|
|
||||||
|
states = torch.stack(env._env.propagate_action_profile(actions), axis=0)
|
||||||
|
assert states.shape == (B, T, nv, 5)
|
||||||
|
|
||||||
|
distance = ((states[:, :, :, :2] - states[:, :, env._agent:env._agent+1, :2])**2).sum(-1).sqrt()
|
||||||
|
distance = torch.where(distance.isnan(), np.inf*torch.ones_like(distance), distance) # only collide with spawned agents
|
||||||
|
distance[:, :, env._agent] = np.inf # cannot collide with itself
|
||||||
|
assert distance.shape == (B, T, nv)
|
||||||
|
|
||||||
|
radius = (env._env._lengths**2 + env._env._widths**2).sqrt() / 2
|
||||||
|
min_distance = radius[env._agent] + radius
|
||||||
|
min_distance = min_distance.unsqueeze(0).unsqueeze(0)
|
||||||
|
assert min_distance.shape == (1, 1, nv)
|
||||||
|
|
||||||
|
collision_tensor = distance > min_distance
|
||||||
|
assert collision_tensor.shape == (B, T, nv)
|
||||||
|
return states, collision_tensor
|
||||||
|
|
||||||
|
def check_future_collisions_fast(env, actions):
|
||||||
|
"""Checks whether `env._agent` would collide with other agents assuming `actions` as input.
|
||||||
|
|
||||||
|
Vehicles are (over-)approximated by single circles.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
env (gym.Env): current environment state
|
||||||
|
actions (list of torch.Tensor): list of B (T, nv, adims) T-length action profiles
|
||||||
|
Returns:
|
||||||
|
feasible (torch.Tensor): tensor of shape (B,) indicating whether the respective action profiles are collision-free
|
||||||
|
"""
|
||||||
|
_, collision_tensor = check_future_collisions_circle(env, actions)
|
||||||
|
return collision_tensor.all(-1).all(-1)
|
||||||
|
|
||||||
|
def check_future_collisions_exact(env, actions):
|
||||||
|
"""
|
||||||
|
Checks whether `env._agent` would collide with other agents assuming `actions` as input.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
env (gym.Env): current environment state
|
||||||
|
actions (list of torch.Tensor): list of B (T, nv, adims) T-length action profiles
|
||||||
|
Returns:
|
||||||
|
feasible (torch.Tensor): tensor of shape (B,) indicating whether the respective action profiles are collision-free
|
||||||
|
"""
|
||||||
|
# First check with simple circle collision check
|
||||||
|
states, collision_tensor = check_future_collisions_circle(env, actions)
|
||||||
|
(B, T, nv, _) = states.shape
|
||||||
|
# For those that have colliding circles, check exactly
|
||||||
|
colliding_mask = ~collision_tensor
|
||||||
|
|
||||||
|
ego_states = states[:, :, env._agent:env._agent+1, :].expand(states.shape)
|
||||||
|
assert ego_states.shape == states.shape
|
||||||
|
|
||||||
|
# get dimensions
|
||||||
|
lengths = env._env._lengths.expand(states.shape[:3])
|
||||||
|
widths = env._env._widths.expand(states.shape[:3])
|
||||||
|
ego_lengths = lengths[:, :, env._agent:env._agent+1].expand(lengths.shape)
|
||||||
|
ego_widths = widths[:, :, env._agent:env._agent+1].expand(widths.shape)
|
||||||
|
assert lengths.shape == widths.shape == ego_lengths.shape == ego_widths.shape == (B, T, nv)
|
||||||
|
|
||||||
|
# For every collision instance between ego and other vehicle, check whether rectangles intersect
|
||||||
|
exact_collisions = torch.zeros_like(collision_tensor[colliding_mask])
|
||||||
|
for i, (ego_state, ego_length, ego_width, other_state, other_length, other_width) in enumerate(zip(
|
||||||
|
ego_states[colliding_mask], ego_lengths[colliding_mask], ego_widths[colliding_mask],
|
||||||
|
states[colliding_mask], lengths[colliding_mask], widths[colliding_mask]
|
||||||
|
)):
|
||||||
|
assert ego_state.shape == other_state.shape == (5,)
|
||||||
|
assert ego_length.shape == ego_width.shape == other_length.shape == other_width.shape == ()
|
||||||
|
p_ego = state_to_polygon(ego_state, ego_length, ego_width)
|
||||||
|
p_other = state_to_polygon(other_state, other_length, other_width)
|
||||||
|
exact_collisions[i] = p_ego.intersects(p_other)
|
||||||
|
|
||||||
|
collision_tensor[colliding_mask] = ~exact_collisions
|
||||||
|
return collision_tensor.all(-1).all(-1)
|
||||||
|
|
||||||
58
src/util/render_env.py
Normal file
58
src/util/render_env.py
Normal file
@@ -0,0 +1,58 @@
|
|||||||
|
|
||||||
|
import stable_baselines3 as sb3
|
||||||
|
from intersim.envs.intersimple import NRasterized
|
||||||
|
|
||||||
|
|
||||||
|
def render_env(model_name='gail_image_multiagent_nocollision', agent=51, environment=NRasterized):
|
||||||
|
"""
|
||||||
|
Render a video from an model, agent, and environment
|
||||||
|
Args:
|
||||||
|
model_name (str): name of the model
|
||||||
|
agent (int): agent to start the video from
|
||||||
|
environment (gym.Env): gym environment class to render environment on
|
||||||
|
"""
|
||||||
|
|
||||||
|
model = sb3.PPO.load(model_name)
|
||||||
|
|
||||||
|
env = environment(stop_on_collision=False, width=36, height=36, m_per_px=2, agent=agent)
|
||||||
|
|
||||||
|
obs = env.reset()
|
||||||
|
i=0
|
||||||
|
while True and i < 600:
|
||||||
|
i+=1
|
||||||
|
action, _states = model.predict(obs)
|
||||||
|
obs, rewards, done, info = env.step(action)
|
||||||
|
env.render(mode='post')
|
||||||
|
if done:
|
||||||
|
break
|
||||||
|
|
||||||
|
env.close(filestr='render/'+model_name+'_agent%i'%(agent))
|
||||||
|
|
||||||
|
def render_options_env(model_name='gail_image_multiagent_nocollision', agent=51, environment=NRasterized):
|
||||||
|
"""
|
||||||
|
Render a video from an model, agent, and environment
|
||||||
|
Args:
|
||||||
|
model_name (str): name of the model
|
||||||
|
agent (int): agent to start the video from
|
||||||
|
environment (gym.Env): gym environment class to render environment on
|
||||||
|
"""
|
||||||
|
|
||||||
|
model = sb3.PPO.load(model_name)
|
||||||
|
|
||||||
|
env = environment(stop_on_collision=False, width=36, height=36, m_per_px=2, agent=agent)
|
||||||
|
|
||||||
|
obs = env.reset()
|
||||||
|
i=0
|
||||||
|
while True and i < 600:
|
||||||
|
i+=1
|
||||||
|
action, _states = model.predict(obs)
|
||||||
|
obs, rewards, done, info = env.step(action)
|
||||||
|
env.render(mode='post')
|
||||||
|
if done:
|
||||||
|
break
|
||||||
|
|
||||||
|
env.close(filestr='render/'+model_name+'_agent%i'%(agent))
|
||||||
|
|
||||||
|
if __name__ == '__main__':
|
||||||
|
import fire
|
||||||
|
fire.Fire(render_env)
|
||||||
45
tests/test_discriminator.py
Normal file
45
tests/test_discriminator.py
Normal file
@@ -0,0 +1,45 @@
|
|||||||
|
from intersim.envs.intersimple import NRasterized
|
||||||
|
from src.discriminator import CnnDiscriminator
|
||||||
|
import torch
|
||||||
|
|
||||||
|
def test_image_concatenation():
|
||||||
|
env = NRasterized()
|
||||||
|
disc = CnnDiscriminator(env)
|
||||||
|
s = torch.tensor(env.reset()).unsqueeze(0)
|
||||||
|
a = torch.tensor([[0.5]])
|
||||||
|
sa = disc._concatenate(s, a)
|
||||||
|
|
||||||
|
assert s.shape == (1, 5, 200, 200)
|
||||||
|
assert a.shape == (1, 1)
|
||||||
|
assert sa.shape == (1, 6, 200, 200)
|
||||||
|
assert torch.allclose(sa[:, :5], 1.0 * s)
|
||||||
|
assert (sa[:, 5] == a.unsqueeze(-1)).all()
|
||||||
|
|
||||||
|
def test_image_concatenation3():
|
||||||
|
env = NRasterized()
|
||||||
|
disc = CnnDiscriminator(env)
|
||||||
|
|
||||||
|
s1 = env.reset()
|
||||||
|
a1 = 0.15
|
||||||
|
s2, _, _, _ = env.step(0.9)
|
||||||
|
a2 = 0.25
|
||||||
|
s3, _, _, _ = env.step(-0.9)
|
||||||
|
a3 = 0.35
|
||||||
|
|
||||||
|
s = torch.stack([
|
||||||
|
torch.tensor(s1),
|
||||||
|
torch.tensor(s2),
|
||||||
|
torch.tensor(s3)
|
||||||
|
], axis=0)
|
||||||
|
a = torch.tensor([
|
||||||
|
[a1],
|
||||||
|
[a2],
|
||||||
|
[a3],
|
||||||
|
])
|
||||||
|
sa = disc._concatenate(s, a)
|
||||||
|
|
||||||
|
assert s.shape == (3, 5, 200, 200)
|
||||||
|
assert a.shape == (3, 1)
|
||||||
|
assert sa.shape == (3, 6, 200, 200)
|
||||||
|
assert torch.allclose(sa[:, :5], 1.0 * s)
|
||||||
|
assert (sa[:, 5] == a.unsqueeze(-1)).all()
|
||||||
Reference in New Issue
Block a user