{"dataset":{"id":"232","dataset_id":"nm000199","name":"Learning from label proportions for a visual matrix speller (ERP)","description":"This dataset comprises event-related potential (ERP) recordings from 13 healthy subjects performing a visual matrix speller task using a calibrationless brain-computer interface approach. The study introduces learning from label proportions (LLP), an unsupervised classification method that exploits known target/non-target stimulus ratios to enable online BCI operation without prior calibration. Subjects performed copy-spelling tasks using a 6×7 character grid across three sessions, achieving 84.5% character accuracy without labeled training data.","owner_user_id":19,"status":"active","github_repo":"nemarDatasets/nm000199","concept_doi":"10.82901/nemar.nm000199","latest_version_doi":"10.82901/nemar.nm000199.v1.0.2","created_at":"2026-03-24 01:06:18","updated_at":"2026-08-18 21:11:18","zenodo_concept_id":"20501013","is_sandbox":0,"visibility":"public","ezid_status":"public","enrichment_json":"{\n  \"version\": \"2.0\",\n  \"pipeline_stage\": \"validated\",\n  \"title\": \"Learning from label proportions for a visual matrix speller (ERP)\",\n  \"description\": \"This dataset comprises event-related potential (ERP) recordings from 13 healthy subjects performing a visual matrix speller task using a calibrationless brain-computer interface approach. The study introduces learning from label proportions (LLP), an unsupervised classification method that exploits known target/non-target stimulus ratios to enable online BCI operation without prior calibration. Subjects performed copy-spelling tasks using a 6×7 character grid across three sessions, achieving 84.5% character accuracy without labeled training data.\",\n  \"methods_description\": \"EEG data were acquired at 1000 Hz from 31 channels using a BrainAmp DC amplifier with passive Ag/AgCl electrodes in a standard 10-20 montage (reference: nose, ground: FCz). Subjects performed a visual P300 speller task with a 6×7 grid, spelling a German pangram sentence three times across three sessions. Two stimulus sequences with different target/non-target ratios were employed: sequence 1 (3 targets/8 stimuli) and sequence 2 (2 targets/18 stimuli). Stimulus onset asynchrony was 250 ms with 100 ms stimulus duration. The LLP classifier was trained online using known class proportions without explicit trial labels.\",\n  \"license\": \"CC-BY-4.0\",\n  \"dataset_type\": \"derivative\",\n  \"authors\": {\n    \"David Hübner\": {\n      \"orcid\": \"0000-0003-4085-9154\"\n    },\n    \"Thibault Verhoeven\": {},\n    \"Konstantin Schmid\": {},\n    \"Klaus-Robert Müller\": {},\n    \"Michael Tangermann\": {},\n    \"Pieter-Jan Kindermans\": {}\n  },\n  \"keywords\": [\n    {\n      \"term\": \"Brain-Computer Interfaces\",\n      \"subject_scheme\": \"MeSH\",\n      \"scheme_uri\": \"https://id.nlm.nih.gov/mesh/\",\n      \"value_uri\": \"http://id.nlm.nih.gov/mesh/D062207\"\n    },\n    {\n      \"term\": \"Event-Related Potentials, P300\",\n      \"subject_scheme\": \"MeSH\",\n      \"scheme_uri\": \"https://id.nlm.nih.gov/mesh/\",\n      \"value_uri\": \"http://id.nlm.nih.gov/mesh/D018913\"\n    },\n    {\n      \"term\": \"P300\"\n    },\n    {\n      \"term\": \"EEG\"\n    },\n    {\n      \"term\": \"unsupervised learning\"\n    },\n    {\n      \"term\": \"visual speller\"\n    },\n    {\n      \"term\": \"learning from label proportions\"\n    }\n  ],\n  \"related_identifiers\": [\n    {\n      \"identifier\": \"10.1371/journal.pone.0175856\",\n      \"identifier_type\": \"DOI\",\n      \"relation_type\": \"IsDerivedFrom\"\n    },\n    {\n      \"identifier\": \"10.5281/zenodo.192684\",\n      \"identifier_type\": \"DOI\",\n      \"relation_type\": \"References\"\n    },\n    {\n      \"identifier\": \"https://github.com/nemarDatasets/nm000199\",\n      \"identifier_type\": \"URL\",\n      \"relation_type\": \"IsDescribedBy\"\n    },\n    {\n      \"identifier\": \"10.21105/joss.01896\",\n      \"identifier_type\": \"DOI\",\n      \"relation_type\": \"References\"\n    },\n    {\n      \"identifier\": \"https://nemar.org/dataset/nm000199\",\n      \"identifier_type\": \"URL\",\n      \"relation_type\": \"IsDescribedBy\"\n    }\n  ],\n  \"funding_references\": [\n    {\n      \"funder_name\": \"Special Research Fund from Ghent University\"\n    },\n    {\n      \"funder_name\": \"DFG\",\n      \"award_number\": \"EXC 1086\",\n      \"award_title\": \"BrainLinks-BrainTools Cluster of Excellence\"\n    },\n    {\n      \"funder_name\": \"bwHPC\",\n      \"award_number\": \"INST 39/963-1 FUGG\"\n    },\n    {\n      \"funder_name\": \"European Union\",\n      \"award_number\": \"657679\",\n      \"award_title\": \"Marie Sklodowska-Curie grant (Horizon 2020)\"\n    },\n    {\n      \"funder_name\": \"Ghent University\",\n      \"award_title\": \"Special Research Fund\"\n    },\n    {\n      \"funder_name\": \"Korean National Research Foundation\",\n      \"award_number\": \"2012-005741\",\n      \"award_title\": \"BK21 program\"\n    }\n  ],\n  \"resource_type_general\": \"Dataset\",\n  \"resource_type_specific\": \"EEG Dataset\",\n  \"modalities\": [\n    \"eeg\"\n  ],\n  \"sizes\": [\n    \"5.5 GB (685 files)\"\n  ],\n  \"formats\": [\n    \".bdf\",\n    \".json\",\n    \".md\",\n    \".tsv\",\n    \".vhdr\",\n    \".yaml\",\n    \".yml\"\n  ],\n  \"source_hash\": \"7fbd18f1c01784f59b56872497a768d5fc38e118ab69014cfb05b9046214d138\"\n}","last_activity_at":"2026-08-16 13:31:29","source":null,"source_id":null,"subject_count":13,"modalities":"eeg","age_min":26,"age_max":26,"file_size":5528956671,"total_files":2595,"tasks":"p300","metadata_columns_error":null,"staleness_warn_stage":null,"staleness_admin_notified_at":null,"authors":"David Hübner, Thibault Verhoeven, Konstantin Schmid, Klaus-Robert Müller, Michael Tangermann, Pieter-Jan Kindermans","license":"CC-BY-4.0","readme":"[![DOI](https://img.shields.io/badge/DOI-10.82901%2Fnemar.nm000199-blue)](https://doi.org/10.82901/nemar.nm000199)\n\n# Learning from label proportions for a visual matrix speller (ERP)\n\nLearning from label proportions for a visual matrix speller (ERP) dataset from Hübner et al 2017 [1]_.\n\n## Dataset Overview\n\n- **Code**: Huebner2017\n- **Paradigm**: p300\n- **DOI**: 10.1371/journal.pone.0175856\n- **Subjects**: 13\n- **Sessions per subject**: 3\n- **Events**: Target=10002, NonTarget=10001\n- **Trial interval**: [-0.2, 0.7] s\n- **Runs per session**: 9\n- **Session IDs**: session_1\n- **File format**: BrainVision\n\n## Acquisition\n\n- **Sampling rate**: 1000.0 Hz\n- **Number of channels**: 31\n- **Channel types**: eeg=31, misc=6\n- **Channel names**: C3, C4, CP1, CP2, CP5, CP6, Cz, EOGvu, F10, F3, F4, F7, F8, F9, FC1, FC2, FC5, FC6, Fp1, Fp2, Fz, O1, O2, P10, P3, P4, P7, P8, P9, Pz, T7, T8, x_EMGl, x_GSR, x_Optic, x_Pulse, x_Respi\n- **Montage**: standard_1020\n- **Hardware**: BrainAmp DC\n- **Reference**: nose\n- **Ground**: FCz\n- **Sensor type**: passive Ag/AgCl\n- **Line frequency**: 50.0 Hz\n- **Impedance threshold**: 20.0 kOhm\n- **Cap manufacturer**: EasyCap\n- **Auxiliary channels**: EOG (1 ch, vertical), pulse, respiration\n\n## Participants\n\n- **Number of subjects**: 13\n- **Health status**: healthy\n- **Age**: mean=26.0, std=1.5\n- **Gender distribution**: female=5, male=8\n- **BCI experience**: mostly naive\n- **Species**: human\n\n## Experimental Protocol\n\n- **Paradigm**: p300\n- **Number of classes**: 2\n- **Class labels**: Target, NonTarget\n- **Trial duration**: 25.0 s\n- **Study design**: Visual ERP speller copy-spelling task using a 6x7 grid with learning from label proportions (LLP) classifier. Two sequences with different target/non-target ratios: sequence 1 (3 targets/8 stimuli), sequence 2 (2 targets/18 stimuli). Unsupervised calibrationless approach.\n- **Feedback type**: visual\n- **Stimulus type**: character matrix\n- **Stimulus modalities**: visual\n- **Primary modality**: visual\n- **Synchronicity**: synchronous\n- **Mode**: online\n- **Training/test split**: False\n- **Instructions**: Copy-spelling task: subjects spelled the sentence 'FRANZY JAGT IM KOMPLETT VERWAHRLOSTEN TAXI QUER DURCH FREIBURG' three times\n- **Stimulus presentation**: soa_ms=250, stimulus_duration_ms=100, grid_size=6x7, highlighting_method=salient (brightness enhancement, rotation, enlargement, trichromatic grid overlay), viewing_distance_cm=80, screen_size_inches=24\n\n## HED Event Annotations\n\nSchema: HED 8.4.0 | Browse: https://www.hedtags.org/hed-schema-browser\n\n```\n  Target\n    ├─ Sensory-event\n    ├─ Experimental-stimulus\n    ├─ Visual-presentation\n    └─ Target\n\n  NonTarget\n    ├─ Sensory-event\n    ├─ Experimental-stimulus\n    ├─ Visual-presentation\n    └─ Non-target\n\n```\n## Paradigm-Specific Parameters\n\n- **Detected paradigm**: p300\n- **Number of targets**: 42\n- **Stimulus onset asynchrony**: 250.0 ms\n\n## Data Structure\n\n- **Trials**: 12852\n- **Trials context**: 68 highlighting events per character, 63 characters per sentence, 3 sentences = 68*63*3 = 12852 EEG epochs per subject. Each epoch is a Target (10002) or NonTarget (10001) event.\n\n## Preprocessing\n\n- **Data state**: raw\n- **Preprocessing applied**: False\n\n## Signal Processing\n\n- **Classifiers**: LLP (Learning from Label Proportions), shrinkage-LDA, EM-algorithm\n- **Feature extraction**: mean amplitude per time interval\n- **Frequency bands**: analyzed=[0.5, 8.0] Hz\n\n## Cross-Validation\n\n- **Method**: 5-fold chronological cross-validation\n- **Folds**: 5\n- **Evaluation type**: within_subject\n\n## Performance (Original Study)\n\n- **Accuracy**: 84.5%\n- **Auc**: 0.975\n- **Online Spelling Accuracy**: 84.5\n- **Post Hoc Spelling Accuracy**: 95.0\n- **Accuracy After Rampup**: 90.2\n- **Supervised Auc**: 0.975\n- **Max Spelling Speed Chars Per Min**: 2.4\n\n## BCI Application\n\n- **Applications**: speller, communication\n- **Environment**: laboratory\n- **Online feedback**: True\n\n## Tags\n\n- **Pathology**: Healthy\n- **Modality**: Visual\n- **Type**: Research\n\n## Documentation\n\n- **DOI**: 10.1371/journal.pone.0175856\n- **License**: CC-BY-4.0\n- **Investigators**: David Hübner, Thibault Verhoeven, Konstantin Schmid, Klaus-Robert Müller, Michael Tangermann, Pieter-Jan Kindermans\n- **Senior author**: Michael Tangermann\n- **Contact**: david.huebner@blbt.uni-freiburg.de; michael.tangermann@blbt.uni-freiburg.de; p.kindermans@tu-berlin.de\n- **Institution**: Albert-Ludwigs-University\n- **Department**: Brain State Decoding Lab, Cluster of Excellence BrainLinks-BrainTools, Department of Computer Science\n- **Address**: Freiburg, Germany\n- **Country**: DE\n- **Repository**: Zenodo\n- **Data URL**: http://doi.org/10.5281/zenodo.192684\n- **Publication year**: 2017\n- **Funding**: BrainLinks-BrainTools Cluster of Excellence funded by the German Research Foundation (DFG), grant number EXC 1086; bwHPC initiative, grant INST 39/963-1 FUGG; European Union's Horizon 2020 research and innovation programme under the Marie Sklodowska-Curie grant agreement No 657679; Special Research Fund from Ghent University; BK21 program funded by Korean National Research Foundation grant No. 2012-005741\n- **Ethics approval**: Ethics Committee of the University Medical Center Freiburg; Declaration of Helsinki\n- **Keywords**: brain-computer interface, BCI, event-related potentials, ERP, P300, learning from label proportions, LLP, unsupervised learning, calibrationless, visual speller\n\n## Abstract\n\nUsing traditional approaches, a brain-computer interface (BCI) requires the collection of calibration data for new subjects prior to online use. This work introduces learning from label proportions (LLP) to the BCI community as a new unsupervised, and easy-to-implement classification approach for ERP-based BCIs. The LLP estimates the mean target and non-target responses based on known proportions of these two classes in different groups of the data. We present a visual ERP speller to meet the requirements of LLP. For evaluation, we ran simulations on artificially created data sets and conducted an online BCI study with 13 subjects performing a copy-spelling task. Theoretical considerations show that LLP is guaranteed to minimize the loss function similar to a corresponding supervised classifier. LLP performed well in simulations and in the online application, where 84.5% of characters were spelled correctly on average without prior calibration.\n\n## Methodology\n\nThe experiment used a modified visual ERP speller with a 6×7 grid. Two distinct stimulus sequences with different target/non-target ratios were used: sequence 1 had 3 targets in 8 stimuli, sequence 2 had 2 targets in 18 stimuli. Each trial consisted of 4 sequences of length 8 and 2 sequences of length 18, totaling 68 highlighting events per character. The LLP algorithm exploited these known proportions to reconstruct mean target and non-target ERP responses without requiring labeled data. The classifier was reset at the start of each sentence and retrained after each character. Subjects spelled a German pangram sentence three times. One subject (S2) had prior EEG experience; others were naive. Sessions lasted about 3 hours including setup. Participants were compensated 8 Euros per hour.\n\n## References\n\nHübner, D., Verhoeven, T., Schmid, K., Müller, K. R., Tangermann, M., & Kindermans, P. J. (2017) Learning from label proportions in brain-computer interfaces: Online unsupervised learning with guarantees. PLOS ONE 12(4): e0175856. https://doi.org/10.1371/journal.pone.0175856\n\n.. versionadded:: 0.4.5\nAppelhoff, S., Sanderson, M., Brooks, T., Vliet, M., Quentin, R., Holdgraf, C., Chaumon, M., Mikulan, E., Tavabi, K., Hochenberger, R., Welke, D., Brunner, C., Rockhill, A., Larson, E., Gramfort, A. and Jas, M. (2019). MNE-BIDS: Organizing electrophysiological data into the BIDS format and facilitating their analysis. Journal of Open Source Software 4: (1896). https://doi.org/10.21105/joss.01896\n\nPernet, C. R., Appelhoff, S., Gorgolewski, K. J., Flandin, G., Phillips, C., Delorme, A., Oostenveld, R. (2019). EEG-BIDS, an extension to the brain imaging data structure for electroencephalography. Scientific Data, 6, 103. https://doi.org/10.1038/s41","bids_version":"1.9.0","sessions_count":3,"publish_date":"2026-06-02 01:44:45","embedding_dirty":0,"license_tier":"attribution","zarr_status":"ready","zarr_converted_at":"2026-09-04 11:02:18","zarr_store_count":342,"zarr_index_etag":"187790fe69b9aa32c44c25cc696deb36","zarr_source_commit":"257ed67de876bfa08b612b198aaba2d50ba98c72","archive_status":"ready","archive_size":4202640038,"archive_retry_count":0,"records_status":"ready","archive_skip_reason":null,"zarr_errors":0,"zarr_failure_count":0,"zarr_deterministic":0,"zarr_failed_at":null,"num_dataset_citations":0,"num_datapaper_citations":46,"n_channels":31,"electrode_system":"10-10","has_hed":1,"hed_version":"8.4.0","is_exemplar":0,"bytes_present":5519920801,"data_complete":1,"withdrawn_at":null,"withdrawn_reason":null,"archive_complete":1,"archive_absent_files":0,"archive_declared_files":2595,"zarr_pool_breaks":0,"total_recording_duration":59228,"recording_duration_min":173,"recording_duration_max":174,"recording_count":342,"recordings_unavailable":0,"recordings_measured":342,"channel_count_min":31,"channel_count_max":31,"sampling_frequency":1000,"power_line_frequency":50,"eeg_reference":"nose","placement_scheme":"10-20 system","sweep_stamps":"{\"enrichment_updated_at\":\"2026-08-18 18:12:38\",\"metadata_updated_at\":\"2026-08-18 21:11:16\",\"archive_checked_at\":\"2026-08-18 21:14:51\",\"zarr_checked_at\":\"2026-06-07 17:58:29\",\"records_checked_at\":\"2026-08-18 21:12:26\",\"citations_updated_at\":\"2026-09-08 03:00:47\",\"channel_montage_checked_at\":\"2026-06-28 22:57:25\",\"hed_checked_at\":\"2026-06-30 07:27:54\",\"data_checked_at\":null,\"availability_report_at\":\"2026-08-20 03:01:29\",\"signal_defaults_at\":\"2026-09-02 11:43:32\",\"recording_stats_at\":\"2026-09-05 03:01:17\"}","participants":13,"num_citations":46,"latest_version":"v1.0.2","zarr_verify_status":null,"zarr_verified_at":null,"owner_username":"bruaristimunha","owner_github":"bruAristimunha","file_size_formatted":"5.15 GB","zarr_data_failures":null,"zarr_index_url":"https://zarr.nemar.org/nm000199/zarr/index.json","attestation_deposit_type":null,"attestation_key_status":null,"attestation_deidentified":null,"attestation_no_duplicate":null,"attestation_upstream_source":null,"attestation_accepted_at":null}}