{"dataset":{"id":"59509","dataset_id":"on004952","name":"ChineseEEG: A Chinese Linguistic Corpora EEG Dataset for Semantic Alignment and Neural Decoding","description":"ChineseEEG is a high-density EEG and simultaneous eye-tracking dataset collected from 10 participants silently reading two Chinese novels over approximately 11 hours. The dataset includes raw and multiple stages of pre-processed EEG data, along with BERT-base-chinese text embeddings of the reading materials, designed to support research on semantic alignment between NLP model representations and neural activity. It provides a resource for studying naturalistic language processing and neural decoding using Chinese linguistic stimuli.","owner_user_id":15,"status":"active","github_repo":"nemarDatasets/on004952","concept_doi":"10.82901/nemar.on004952","latest_version_doi":"10.82901/nemar.on004952.v1.0.0","created_at":"2026-06-26 00:31:54","updated_at":"2026-08-19 02:16:39","zenodo_concept_id":null,"is_sandbox":0,"visibility":"public","ezid_status":"public","enrichment_json":"{\n  \"version\": \"2.0\",\n  \"pipeline_stage\": \"validated\",\n  \"title\": \"ChineseEEG: A Chinese Linguistic Corpora EEG Dataset for Semantic Alignment and Neural Decoding\",\n  \"description\": \"ChineseEEG is a high-density EEG and simultaneous eye-tracking dataset collected from 10 participants silently reading two Chinese novels over approximately 11 hours. The dataset includes raw and multiple stages of pre-processed EEG data, along with BERT-base-chinese text embeddings of the reading materials, designed to support research on semantic alignment between NLP model representations and neural activity. It provides a resource for studying naturalistic language processing and neural decoding using Chinese linguistic stimuli.\",\n  \"methods_description\": \"EEG data were recorded at 1 kHz sampling rate using high-density EEG (EGI system, .mff format converted to BrainVision), alongside simultaneous eye-tracking data, while participants silently read Chinese novels presented in a moving-window highlighted reading paradigm. Data were pre-processed through filtering (0.5-80 Hz and 0.5-30 Hz band-pass) and further processed with a full pipeline including bad channel rejection and ICA, down-sampled to 256 Hz. The dataset was organized following the EEG-BIDS specification using the MNE-BIDS package.\",\n  \"license\": \"CC0\",\n  \"dataset_type\": \"raw\",\n  \"authors\": {\n    \"Xinyu Mou\": {\n      \"orcid\": \"0009-0004-3273-9704\"\n    },\n    \"Cuilin He\": {},\n    \"Liwei Tan\": {},\n    \"Junjie Yu\": {},\n    \"Huadong Liang\": {},\n    \"Jianyu Zhang\": {},\n    \"Tian Yan\": {},\n    \"Yu-Fang Yang\": {},\n    \"Ting Xu\": {\n      \"orcid\": \"0000-0002-0065-3832\"\n    },\n    \"Qing Wang\": {},\n    \"Miao Cao\": {},\n    \"Zijiao Chen\": {},\n    \"Chuan-Peng Hu\": {\n      \"orcid\": \"0000-0002-7503-5131\"\n    },\n    \"Xindi Wang\": {},\n    \"Quanying Liu\": {\n      \"orcid\": \"0000-0002-2501-7656\"\n    },\n    \"Haiyan Wu\": {\n      \"orcid\": \"0000-0001-8869-6636\"\n    }\n  },\n  \"keywords\": [\n    {\n      \"term\": \"EEG\"\n    },\n    {\n      \"term\": \"Eye-Tracking\"\n    },\n    {\n      \"term\": \"Reading\",\n      \"subject_scheme\": \"MeSH\",\n      \"value_uri\": \"http://id.nlm.nih.gov/mesh/D011932\"\n    },\n    {\n      \"term\": \"Semantics\",\n      \"subject_scheme\": \"MeSH\",\n      \"value_uri\": \"http://id.nlm.nih.gov/mesh/D012660\"\n    },\n    {\n      \"term\": \"Natural Language Processing\"\n    },\n    {\n      \"term\": \"Chinese language\"\n    },\n    {\n      \"term\": \"Neural decoding\"\n    }\n  ],\n  \"related_identifiers\": [\n    {\n      \"identifier\": \"10.1038/s41597-024-03398-7\",\n      \"identifier_type\": \"DOI\",\n      \"relation_type\": \"IsDescribedBy\"\n    },\n    {\n      \"identifier\": \"https://github.com/nemarDatasets/on004952\",\n      \"identifier_type\": \"URL\",\n      \"relation_type\": \"IsDescribedBy\"\n    },\n    {\n      \"identifier\": \"10.18112/openneuro.ds004952.v1.2.2\",\n      \"identifier_type\": \"DOI\",\n      \"relation_type\": \"IsDerivedFrom\"\n    },\n    {\n      \"identifier\": \"https://nemar.org/dataset/on004952\",\n      \"identifier_type\": \"URL\",\n      \"relation_type\": \"IsDescribedBy\"\n    },\n    {\n      \"identifier\": \"10.1101/2024.02.08.579481\",\n      \"identifier_type\": \"DOI\",\n      \"relation_type\": \"References\"\n    }\n  ],\n  \"resource_type_general\": \"Dataset\",\n  \"resource_type_specific\": \"EEG Dataset\",\n  \"modalities\": [\n    \"eeg\"\n  ],\n  \"sizes\": [\n    \"748.0 GB (3644 files)\"\n  ],\n  \"formats\": [\n    \".db\",\n    \".eeg\",\n    \".json\",\n    \".md\",\n    \".npy\",\n    \".png\",\n    \".rar\",\n    \".tsv\",\n    \".txt\",\n    \".vhdr\",\n    \".vmrk\",\n    \".xlsx\",\n    \".yml\"\n  ],\n  \"source_hash\": \"d483a933cb3f1f94f5dc5160ce89ee7784ab9578ff0d4ac4c0cbdab8a2926fa7\"\n}","last_activity_at":"2026-06-26 00:31:54","source":"openneuro","source_id":"ds004952","subject_count":10,"modalities":"eeg","age_min":null,"age_max":null,"file_size":748095807833,"total_files":10856,"tasks":"reading","metadata_columns_error":null,"staleness_warn_stage":null,"staleness_admin_notified_at":null,"authors":"Xinyu Mou, Cuilin He, Liwei Tan, Junjie Yu, Huadong Liang, Jianyu Zhang, Tian Yan, Yu-Fang Yang, Ting Xu, Qing Wang, Miao Cao, Zijiao Chen, Chuan-Peng Hu, Xindi Wang, Quanying Liu, Haiyan Wu","license":"CC0","readme":"[![DOI](https://img.shields.io/badge/DOI-10.82901%2Fnemar.on004952-blue)](https://doi.org/10.82901/nemar.on004952)\n\n# ChineseEEG: A Chinese Linguistic Corpora EEG Dataset for Semantic Alignment and Neural Decoding\n\n## Introduction\n\n\"ChineseEEG\" (Chinese Linguistic Corpora EEG Dataset) contains high-density EEG data and simultaneous eye-tracking data recorded from 10 participants, each silently reading Chinese text for about 11 hours. This dataset further comprises pre-processed EEG sensor-level data generated under different parameter settings, offering researchers a diverse range of selections. Additionally, we provide embeddings of the Chinese text materials encoded from BERT-base-chinese model, which is a pre-trained NLP specifically used for Chinese, aiding researchers in exploring the alignment between text embeddings from NLP models and brain information representations.\n\n## Participant Overview\n\nIn total, data from 10 participants were used (18-24 years old, averaged 20.68 years old, and 5 males). No participants reported neurological or psychiatric history. All participants are right-handed and have normal or corrected-to-normal vision. \n\n## Experiment Materials\n\nThe experimental materials consist of two novels in Chinese, both in the genre of children's literature. The first is **The Little Prince** and the second is **Garnett Dream**. \nFor **The Little Prince**, the preface was used as material for the practice reading phase. The main body of the novel was then used for seven sessions in the formal reading phase. The first six sessions each included 4 chapters of the novel, while the seventh session included the last two chapters.\nFor **Garnett Dream**, the first 18 chapters were used for 18 sessions in the formal reading stage, with each session including a complete chapter. \n\nTo properly present the text on the screen during the experiments, the content of each session was segmented into a series of units, with each unit containing no more than 10 Chinese characters. These segmented contents were saved in Excel (.xlsx) format for subsequent usage. During the experiment, three adjacent units from each session's content will be displayed on the screen in three separate lines, with the middle line highlighted for the participant to read. \nIn summary, a total of 115,233 characters (24,324 in **The Little Prince** and 90,909 in **Garnett Dream**), of which 2985 characters were unique, were used as experimental stimuli in ChineseEEG dataset. \n\nThe original and segmented novels are saved in the `derivatives/novels` folder. The `segmented_novel` folder in `novels` folder contains two types of Excel files: one type of file has names ending with \"display,\" while the other type does not contain this suffix. The former stores units that have been segmented; the latter includes units that have been reassembled according to the experimental presentation format. These files ending with \"display\" will be used to support the execution of relevant code, in order to achieve effective stimulus presentation in the experiment. \n\nThe code for generating these two types of files, as well as the code for experimental presentation, can be found in the GitHub repository: https://github.com/ncclabsustech/Chinese_reading_task_eeg_processing.\n\n## Experiment Procedures\n\nParticipants were tasked with reading a novel and were required to keep their heads still and keep their gaze on the highlighted (red) Chinese characters moving across the screen, reading at a pace set by the program. They were required to read an entire novel in multiple runs within a single session. Each run is divided into two phases: the eye-tracker calibration phase and the reading phase.\n\nThe eye-tracker calibration phase is at the beginning of each run, requiring participants to keep their gaze at a fixation point, which sequentially appeared at the four corners and the center of the screen.\n\nIn the reading phase, the screen initially displayed the serial number of the current chapter. Subsequently, the text appeared with three lines per page, ensuring each line contained no more than ten Chinese characters (excluding punctuation). On each page, the middle line was highlighted as the focal point, while the upper and lower lines were displayed with reduced intensity as the background. Each character in the middle line was sequentially highlighted with red color for 0.35 s, and participants were required to read the novel content following the highlighted cues.\n\nFor detailed information about the experiment settings and procedures, please refer to our paper at https://doi.org/10.1101/2024.02.08.579481.\n\n## Markers\n\nTo precisely co-register EEG segments with individual characters during the experiment, we marked the EEG data with triggers. \n\n- EYES: Eyetracker starts to record\n- EYEE: Eyetracker stops recording\n- CALS: Eyetracker calibration starts\n- CALE: Eyetracker calibration stops\n- BEGN: EGI starts to record\n- STOP: EGI stops recording\n- CHxx：Beginning of specific chapter (Numbers correspond with chapters) \n- ROWS: Beginning of a row\n- ROWE: End of a row\n- PRES：Beginning of the preface\n- PREE：End of the preface\n\n## Data Record\n\nThe raw EEG data has a sampling rate of 1 kHz, while the filtered data and pre-processed data has a sampling rate of 256 Hz.\n\n### Data Structure\n\nThe dataset is organized following the EEG-BIDS specification using the MNE-BIDS package. The dataset contains some regular BIDS files, 10 participants’ data folders, and a derivatives folder. The stand-alone files offer an overview of the dataset: i) dataset_description.json is a JSON file depicting the information of the dataset, such as the name, dataset type and authors; ii) participants.tsv contains participants’ information, such as age, sex, and handedness; iii) participants.json describes the column attributes in participants.tsv; iv) README.md contains a detailed introduction of the dataset.\nEach participant’s folder contains two folders named ses-LittlePrince and ses-GarnettDream, which store the data of this\nparticipant reading two novels, respectively. Each of the two folders contains a folder eeg and one file sub-xx_scans.tsv. The tsv\nfile contains information about the scanning time of each file. The eeg folder contains the source raw EEG data of several runs,\nchannels, and marker events files. Each run includes an eeg.json file, which encompasses detailed information for that run,\nsuch as the sampling rate and the number of channels. Events are stored in events.tsv with onset and event ID. The EEG data\nis converted from raw metafile format (.mff file) to BrainVision format (.vhdr, .vmrk and .eeg files) since EEG-BIDS is not\nofficially compatible with .mff format.\nThe derivatives folder contains six folders: eyetracking_data, filtered_0.5_80, filtered_0.5_30, preproc, novels, and text_embeddings. The eyetracking_data folder contains all the eye-tracking data. Each eye-tracking data is formatted in a\n.zip file with eye moving trajectories and other parameters like sampling rate saved in different files. The filtered_0.5_80\nfolder and filtered_0.5_30 folder contain data that has been processed up to the pre-processing step of 0.5-80 Hz and 0.5-30\nHz band-pass filtering respectively. This data is suitable for researchers who have specific requirements and want to perform\ncustomized processing on subsequent pre-processing steps like ICA and re-referencing. The preproc folder contains minimally\npre-processed EEG data that is processed using the whole pre-processing pipeline. It includes four additional types of files\ncompared to the participants’ raw data folders in the root directory: i) bad_channels.json contains bad channels marked during\nbad channel rejection phase. ii) ica_components.npy stores the values of all independent components in the ICA phase. iii)\nica_components.json includes the independent components excluded in ICA (the ICA random seed is fixed, allowing for\nreproducible results). iv) ica_components_topography.png is a picture of the topographic maps of all independent components,\nwhere the excluded components are labeled in grey. The novels folder contains the original and segmented tex","bids_version":"1.7.0","sessions_count":2,"publish_date":null,"embedding_dirty":0,"license_tier":"public","zarr_status":"ready","zarr_converted_at":"2026-08-20 23:10:17","zarr_store_count":319,"zarr_index_etag":"26a190498336a858b729c4a385f2dabc","zarr_source_commit":"13231a31a02b329c61aff32d953643424ebc2fed","archive_status":null,"archive_size":null,"archive_retry_count":0,"records_status":"ready","archive_skip_reason":"dataset 696.7 GB exceeds 100.0 GB archive limit; use direct download","zarr_errors":877,"zarr_failure_count":877,"zarr_deterministic":1,"zarr_failed_at":"2026-08-20 23:10:17","num_dataset_citations":1,"num_datapaper_citations":23,"n_channels":128,"electrode_system":"egi-geodesic","has_hed":0,"hed_version":null,"is_exemplar":0,"bytes_present":748035227886,"data_complete":1,"withdrawn_at":null,"withdrawn_reason":null,"archive_complete":null,"archive_absent_files":null,"archive_declared_files":null,"zarr_pool_breaks":null,"total_recording_duration":443816.2199999999,"recording_duration_min":682.916,"recording_duration_max":1984.2,"recording_count":1196,"recordings_unavailable":877,"recordings_measured":319,"channel_count_min":128,"channel_count_max":128,"sampling_frequency":1000,"power_line_frequency":50,"eeg_reference":null,"placement_scheme":null,"sweep_stamps":"{\"enrichment_updated_at\":\"2026-08-19 02:16:26\",\"metadata_updated_at\":\"2026-08-19 02:16:38\",\"archive_checked_at\":\"2026-06-26 01:10:13\",\"zarr_checked_at\":null,\"records_checked_at\":\"2026-06-26 01:10:51\",\"citations_updated_at\":\"2026-09-08 03:00:48\",\"channel_montage_checked_at\":\"2026-06-28 23:36:08\",\"hed_checked_at\":\"2026-06-30 05:09:18\",\"data_checked_at\":null,\"availability_report_at\":\"2026-07-23 01:22:01\",\"recording_stats_at\":\"2026-09-02 11:33:04\",\"signal_defaults_at\":\"2026-09-02 12:23:00\"}","participants":10,"num_citations":24,"latest_version":"v1.0.0","zarr_verify_status":null,"zarr_verified_at":null,"owner_username":"nemarAdmin","owner_github":"nemarAdmin","file_size_formatted":"697 GB","zarr_data_failures":{"count":877,"detail_ref":"zarr/index.json","compacted_by":"migration_0074"},"zarr_index_url":"https://zarr.nemar.org/on004952/zarr/index.json","attestation_deposit_type":null,"attestation_key_status":null,"attestation_deidentified":null,"attestation_no_duplicate":null,"attestation_upstream_source":null,"attestation_accepted_at":null}}