File size: 1,997 Bytes
6073e55
23fdbc0
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
# Copyright (c) 2025 Ye Liu. Licensed under the BSD-3-Clause License.

import csv

import nncore

from videomind.dataset.hybrid import DATASETS
from videomind.dataset.wrappers import AnsweringDataset
from videomind.utils.parser import parse_query, parse_question


@DATASETS.register(name='nextqa')
class NExTQADataset(AnsweringDataset):

    ANNO_PATH_TRAIN = 'data/nextqa/train.csv'
    ANNO_PATH_VALID = 'data/nextqa/val.csv'
    ANNO_PATH_TEST = 'data/nextqa/test.csv'

    VIDEO_ID_MAP = 'data/nextqa/map_vid_vidorID.json'
    VIDEO_ROOT = 'data/nextqa/NExTVideo'

    @classmethod
    def load_annos(self, split='train'):
        if split == 'train':
            anno_path = self.ANNO_PATH_TRAIN
        elif split == 'valid':
            anno_path = self.ANNO_PATH_VALID
        else:
            anno_path = self.ANNO_PATH_TEST

        with open(anno_path, mode='r') as f:
            reader = csv.DictReader(f)
            raw_annos = [d for d in reader]

        video_id_map = nncore.load(self.VIDEO_ID_MAP)

        annos = []
        for raw_anno in raw_annos:
            vid = raw_anno['video']
            qid = raw_anno['qid']

            video_id = video_id_map[vid]
            query = parse_query(raw_anno['question'].capitalize() + '?')
            question = parse_question(raw_anno['question'].capitalize() + '?')
            options = [raw_anno[k].capitalize() for k in ('a0', 'a1', 'a2', 'a3', 'a4')]
            ans = chr(ord('A') + int(raw_anno['answer']))
            answer = options[int(raw_anno['answer'])]

            anno = dict(
                source='nextqa',
                data_type='multimodal',
                uid=f'{vid}_{qid}',
                video_path=nncore.join(self.VIDEO_ROOT, video_id + '.mp4'),
                query=query,
                question=question,
                options=options,
                answer=answer,
                ans=ans,
                task=raw_anno['type'])

            annos.append(anno)

        return annos