{
  "paper": {
    "title": "ReViV: Reconstructing the Viewer and the View in 4D from Monocular Egocentric Video",
    "source": "https://arxiv.org/html/2607.17790",
    "version": "arXiv:2607.17790v1, submitted 2026-07-20",
    "complete_source": true,
    "source_notes": "The complete arXiv HTML and ar5iv renderings, including appendices, were cross-checked. The 35.3 MB PDF endpoint was too slow to complete in the available network session, so no claim relies on the partial PDF file."
  },
  "units": [
    {"id":"source-status","kind":"source","source_ref":"arXiv source package status","status":"covered","article_anchor":"source-status","notes":""},
    {"id":"source-pdf","kind":"source","source_ref":"arXiv PDF v1","status":"unavailable","article_anchor":"source-pdf","notes":"The PDF transfer repeatedly timed out; the complete arXiv HTML and independent ar5iv rendering were used instead, and the incomplete PDF was not treated as evidence."},
    {"id":"abs","kind":"abstract","source_ref":"Abstract","status":"covered","article_anchor":"abs","notes":""},
    {"id":"sec-1","kind":"section","source_ref":"Section 1 Introduction","status":"covered","article_anchor":"sec-1","notes":""},
    {"id":"fig-1","kind":"figure","source_ref":"Figure 1","status":"covered","article_anchor":"fig-1","notes":""},
    {"id":"claim-1","kind":"contribution","source_ref":"Introduction contribution 1","status":"covered","article_anchor":"claim-1","notes":""},
    {"id":"claim-2","kind":"contribution","source_ref":"Introduction contribution 2","status":"covered","article_anchor":"claim-2","notes":""},
    {"id":"claim-3","kind":"contribution","source_ref":"Introduction contribution 3","status":"covered","article_anchor":"claim-3","notes":""},
    {"id":"sec-2","kind":"section","source_ref":"Section 2 Related Work","status":"covered","article_anchor":"sec-2","notes":""},
    {"id":"sec-2-1","kind":"subsection","source_ref":"Section 2.1 4D Reconstruction","status":"covered","article_anchor":"sec-2-1","notes":""},
    {"id":"sec-2-2","kind":"subsection","source_ref":"Section 2.2 Scene Reconstruction from Egocentric Videos","status":"covered","article_anchor":"sec-2-2","notes":""},
    {"id":"sec-2-3","kind":"subsection","source_ref":"Section 2.3 Human Motion Estimation from Egocentric Videos","status":"covered","article_anchor":"sec-2-3","notes":""},
    {"id":"sec-3","kind":"section","source_ref":"Section 3 Data Engine","status":"covered","article_anchor":"sec-3","notes":""},
    {"id":"tab-1","kind":"table","source_ref":"Table 1","status":"covered","article_anchor":"tab-1","notes":""},
    {"id":"sec-3-1","kind":"subsection","source_ref":"Section 3.1 Temporally Consistent Geometric Pseudo-Labeling","status":"covered","article_anchor":"sec-3-1","notes":""},
    {"id":"sec-3-2","kind":"subsection","source_ref":"Section 3.2 Unified Kinematic Representation","status":"covered","article_anchor":"sec-3-2","notes":""},
    {"id":"fig-2","kind":"figure","source_ref":"Figure 2","status":"covered","article_anchor":"fig-2","notes":""},
    {"id":"sec-4","kind":"section","source_ref":"Section 4 Method","status":"covered","article_anchor":"sec-4","notes":""},
    {"id":"sec-4-1","kind":"subsection","source_ref":"Section 4.1 Problem Formulation","status":"covered","article_anchor":"sec-4-1","notes":""},
    {"id":"eq-1","kind":"equation","source_ref":"Equation (1)","status":"covered","article_anchor":"eq-1","notes":""},
    {"id":"metric-alignment","kind":"method","source_ref":"Section 4.1 Metric Alignment of 4D Reconstruction","status":"covered","article_anchor":"metric-alignment","notes":""},
    {"id":"sec-4-2","kind":"subsection","source_ref":"Section 4.2 Unified Discrete Representation","status":"covered","article_anchor":"sec-4-2","notes":""},
    {"id":"def-vq","kind":"definition","source_ref":"Section 4.2 modality-specific quantization definition","status":"covered","article_anchor":"def-vq","notes":""},
    {"id":"eq-2","kind":"equation","source_ref":"Equation (2)","status":"covered","article_anchor":"eq-2","notes":""},
    {"id":"sec-4-3","kind":"subsection","source_ref":"Section 4.3 Masked Generative Egocentric Transformer","status":"covered","article_anchor":"sec-4-3","notes":""},
    {"id":"eq-3","kind":"equation","source_ref":"Equation (3)","status":"covered","article_anchor":"eq-3","notes":""},
    {"id":"sec-5","kind":"section","source_ref":"Section 5 Experiment","status":"covered","article_anchor":"sec-5","notes":""},
    {"id":"sec-5-1","kind":"subsection","source_ref":"Section 5.1 Egocentric Body Motion Reconstruction","status":"covered","article_anchor":"sec-5-1","notes":""},
    {"id":"tab-2","kind":"table","source_ref":"Table 2","status":"covered","article_anchor":"tab-2","notes":""},
    {"id":"exp-body","kind":"experiment","source_ref":"Body motion protocol, metrics and results","status":"covered","article_anchor":"exp-body","notes":""},
    {"id":"sec-5-2","kind":"subsection","source_ref":"Section 5.2 Egocentric Hand Motion Reconstruction","status":"covered","article_anchor":"sec-5-2","notes":""},
    {"id":"tab-3","kind":"table","source_ref":"Table 3","status":"covered","article_anchor":"tab-3","notes":""},
    {"id":"fig-3","kind":"figure","source_ref":"Figure 3","status":"covered","article_anchor":"fig-3","notes":""},
    {"id":"exp-hand","kind":"experiment","source_ref":"Hand motion protocol, metrics and results","status":"covered","article_anchor":"exp-hand","notes":""},
    {"id":"sec-5-3","kind":"subsection","source_ref":"Section 5.3 Egocentric Camera Tracking","status":"covered","article_anchor":"sec-5-3","notes":""},
    {"id":"tab-4","kind":"table","source_ref":"Table 4","status":"covered","article_anchor":"tab-4","notes":""},
    {"id":"exp-camera","kind":"experiment","source_ref":"Camera tracking protocol and results","status":"covered","article_anchor":"exp-camera","notes":""},
    {"id":"sec-5-4","kind":"subsection","source_ref":"Section 5.4 Egocentric Gaze Estimation","status":"covered","article_anchor":"sec-5-4","notes":""},
    {"id":"exp-gaze","kind":"experiment","source_ref":"Gaze estimation protocol and results","status":"covered","article_anchor":"exp-gaze","notes":""},
    {"id":"sec-5-5","kind":"subsection","source_ref":"Section 5.5 Egocentric Video Depth Estimation","status":"covered","article_anchor":"sec-5-5","notes":""},
    {"id":"exp-depth","kind":"experiment","source_ref":"Depth estimation protocol and results","status":"covered","article_anchor":"exp-depth","notes":""},
    {"id":"sec-5-6","kind":"subsection","source_ref":"Section 5.6 Ablation Studies","status":"covered","article_anchor":"sec-5-6","notes":""},
    {"id":"tab-5","kind":"table","source_ref":"Table 5","status":"covered","article_anchor":"tab-5","notes":""},
    {"id":"exp-ablation-main","kind":"experiment","source_ref":"Main ablation studies","status":"covered","article_anchor":"exp-ablation-main","notes":""},
    {"id":"sec-6","kind":"section","source_ref":"Section 6 Conclusion","status":"covered","article_anchor":"sec-6","notes":""},
    {"id":"lim-1","kind":"limitation","source_ref":"Limitations and Future Directions","status":"covered","article_anchor":"lim-1","notes":""},
    {"id":"app-a","kind":"appendix","source_ref":"Appendix 0.A Tokenizer Details","status":"covered","article_anchor":"app-a","notes":""},
    {"id":"tab-6","kind":"table","source_ref":"Table 6","status":"covered","article_anchor":"tab-6","notes":""},
    {"id":"app-b","kind":"appendix","source_ref":"Appendix 0.B Masked Generative Egocentric Transformer Details","status":"covered","article_anchor":"app-b","notes":""},
    {"id":"alg-1","kind":"algorithm","source_ref":"Algorithm 1","status":"covered","article_anchor":"alg-1","notes":""},
    {"id":"exp-training","kind":"experiment-setting","source_ref":"Appendix 0.B MGET training details","status":"covered","article_anchor":"exp-training","notes":""},
    {"id":"app-c","kind":"appendix","source_ref":"Appendix 0.C Inference Details","status":"covered","article_anchor":"app-c","notes":""},
    {"id":"eq-cfg","kind":"equation-group","source_ref":"Appendix 0.C three-step decoding and classifier-free guidance","status":"covered","article_anchor":"eq-cfg","notes":""},
    {"id":"app-d","kind":"appendix","source_ref":"Appendix 0.D Ego-body anchored depth scale optimization","status":"covered","article_anchor":"app-d","notes":""},
    {"id":"eq-depth-scale","kind":"equation-group","source_ref":"Appendix 0.D least-squares depth scale alignment","status":"covered","article_anchor":"eq-depth-scale","notes":""},
    {"id":"app-e","kind":"appendix","source_ref":"Appendix 0.E Additional Experiment Results","status":"covered","article_anchor":"app-e","notes":""},
    {"id":"app-e-1","kind":"subsection","source_ref":"Appendix 0.E.1 Egocentric Camera Tracking","status":"covered","article_anchor":"app-e-1","notes":""},
    {"id":"fig-4","kind":"figure","source_ref":"Figure 4","status":"covered","article_anchor":"fig-4","notes":""},
    {"id":"fig-5","kind":"figure","source_ref":"Figure 5","status":"covered","article_anchor":"fig-5","notes":""},
    {"id":"app-e-2","kind":"subsection","source_ref":"Appendix 0.E.2 Egocentric Gaze Estimation","status":"covered","article_anchor":"app-e-2","notes":""},
    {"id":"fig-6","kind":"figure","source_ref":"Figure 6","status":"covered","article_anchor":"fig-6","notes":""},
    {"id":"app-e-3","kind":"subsection","source_ref":"Appendix 0.E.3 Egocentric Body Motion Reconstruction","status":"covered","article_anchor":"app-e-3","notes":""},
    {"id":"fig-7","kind":"figure","source_ref":"Figure 7","status":"covered","article_anchor":"fig-7","notes":""},
    {"id":"app-f","kind":"appendix","source_ref":"Appendix 0.F Additional Ablation Studies","status":"covered","article_anchor":"app-f","notes":""},
    {"id":"app-f-1","kind":"subsection","source_ref":"Appendix 0.F.1 Egocentric Hand Motion Reconstruction Ablation","status":"covered","article_anchor":"app-f-1","notes":""},
    {"id":"tab-7","kind":"table","source_ref":"Table 7","status":"covered","article_anchor":"tab-7","notes":""},
    {"id":"app-f-2","kind":"subsection","source_ref":"Appendix 0.F.2 Vision Transformer Encoding Branch","status":"covered","article_anchor":"app-f-2","notes":""},
    {"id":"tab-8","kind":"table","source_ref":"Table 8","status":"covered","article_anchor":"tab-8","notes":""},
    {"id":"app-f-3","kind":"subsection","source_ref":"Appendix 0.F.3 VQ-VAE Ablation","status":"covered","article_anchor":"app-f-3","notes":""},
    {"id":"exp-vq-ablation","kind":"experiment","source_ref":"Appendix 0.F.3 codebook and tubelet ablation","status":"covered","article_anchor":"exp-vq-ablation","notes":""},
    {"id":"code-tokenizers","kind":"code-mapping","source_ref":"Official code: motion and visual tokenizers","status":"covered","article_anchor":"code-tokenizers","notes":""},
    {"id":"code-masking","kind":"code-mapping","source_ref":"Official code: Dirichlet multimodal masking","status":"covered","article_anchor":"code-masking","notes":""},
    {"id":"code-model","kind":"code-mapping","source_ref":"Official code: MGET encoder-decoder attention","status":"covered","article_anchor":"code-model","notes":""},
    {"id":"code-inference","kind":"code-mapping","source_ref":"Official code: released inference pathways and schedule","status":"covered","article_anchor":"code-inference","notes":""},
    {"id":"refs-main","kind":"bibliography","source_ref":"Main References","status":"not-applicable","article_anchor":"refs-main","notes":"The reference list is not repeated item by item; all works used to define the paper's technical position are covered in Section 2."},
    {"id":"refs-supp","kind":"bibliography","source_ref":"Supplementary References","status":"not-applicable","article_anchor":"refs-supp","notes":"The supplementary reference list is not repeated item by item; cited implementation and comparison anchors are identified in the relevant appendix sections."}
  ]
}
