simplify pre-training dataset, use npz

This commit is contained in:
allenzren
2024-09-08 17:52:16 -04:00
parent 447c8dfd02
commit 8ce0aa1485
66 changed files with 170 additions and 324 deletions
+3
View File
@@ -124,6 +124,9 @@ class VisionDiffusionMLP(nn.Module):
else:
state = cond["state"]
# convert rgb to float32 for augmentation
rgb = rgb.float()
# get vit output - pass in two images separately
if rgb.shape[1] == 6: # TODO: properly handle multiple images
rgb1 = rgb[:, :3]