{"dataset":{"id":"228","dataset_id":"nm000195","name":"Mixture of LLP and EM for a visual matrix speller (ERP) dataset from","description":"This dataset comprises EEG recordings from a P300 visual matrix speller study comparing three unsupervised learning methods (Expectation-Maximization, Learning from Label Proportions, and their combination MIX) for brain-computer interface decoding. Twelve healthy participants performed a copy-spelling task using a modified 6×6 character grid extended with 10 # symbols as visual blanks (46 total symbols), recorded at 1000 Hz from 31 EEG channels. The study demonstrates that unsupervised learning methods can achieve performance comparable to supervised approaches without requiring calibration data.","owner_user_id":19,"status":"active","github_repo":"nemarDatasets/nm000195","concept_doi":"10.82901/nemar.nm000195","latest_version_doi":"10.82901/nemar.nm000195.v1.0.2","created_at":"2026-03-24 00:43:21","updated_at":"2026-08-18 21:09:54","zenodo_concept_id":"20500863","is_sandbox":0,"visibility":"public","ezid_status":"public","enrichment_json":"{\n  \"version\": \"2.0\",\n  \"pipeline_stage\": \"enriched\",\n  \"title\": \"Mixture of LLP and EM for a visual matrix speller (ERP) dataset from\",\n  \"description\": \"This dataset comprises EEG recordings from a P300 visual matrix speller study comparing three unsupervised learning methods (Expectation-Maximization, Learning from Label Proportions, and their combination MIX) for brain-computer interface decoding. Twelve healthy participants performed a copy-spelling task using a modified 6×6 character grid extended with 10 # symbols as visual blanks (46 total symbols), recorded at 1000 Hz from 31 EEG channels. The study demonstrates that unsupervised learning methods can achieve performance comparable to supervised approaches without requiring calibration data.\",\n  \"methods_description\": \"Online study with 12 healthy participants (8 female, 4 male; mean age 26 years, range 19-31) performing a P300 copy-spelling task. Participants spelled a 35-character German sentence across three blocks, each using a different unsupervised algorithm (EM, LLP, or MIX) in pseudo-randomized order. EEG was recorded at 1000 Hz from 31 channels (Ag/AgCl electrodes in extended 10-20 montage) using BrainAmp DC amplifier with nose reference. Visual stimuli were presented on a 24-inch screen at 80 cm distance with 250 ms stimulus onset asynchrony and 100 ms stimulus duration. A modified 6×6 grid with 10 additional # symbols was used with two interleaved highlighting sequences to enable unsupervised learning.\",\n  \"license\": \"CC-BY-4.0\",\n  \"dataset_type\": \"derivative\",\n  \"authors\": {\n    \"David Hübner\": {},\n    \"Thibault Verhoeven\": {},\n    \"Klaus-Robert Müller\": {},\n    \"Pieter-Jan Kindermans\": {},\n    \"Michael Tangermann\": {}\n  },\n  \"keywords\": [\n    {\n      \"term\": \"Brain-Computer Interfaces\",\n      \"subject_scheme\": \"MeSH\",\n      \"scheme_uri\": \"https://id.nlm.nih.gov/mesh/\",\n      \"value_uri\": \"http://id.nlm.nih.gov/mesh/D062207\"\n    },\n    {\n      \"term\": \"Event-Related Potentials, P300\",\n      \"subject_scheme\": \"MeSH\",\n      \"scheme_uri\": \"https://id.nlm.nih.gov/mesh/\",\n      \"value_uri\": \"http://id.nlm.nih.gov/mesh/D018913\"\n    },\n    {\n      \"term\": \"EEG\"\n    },\n    {\n      \"term\": \"P300 speller\"\n    },\n    {\n      \"term\": \"matrix speller\"\n    },\n    {\n      \"term\": \"unsupervised learning\"\n    },\n    {\n      \"term\": \"expectation-maximization\"\n    },\n    {\n      \"term\": \"learning from label proportions\"\n    },\n    {\n      \"term\": \"hybrid learning method\"\n    }\n  ],\n  \"related_identifiers\": [\n    {\n      \"identifier\": \"10.1109/MCI.2018.2807039\",\n      \"identifier_type\": \"DOI\",\n      \"relation_type\": \"IsDerivedFrom\"\n    },\n    {\n      \"identifier\": \"https://github.com/nemarDatasets/nm000195\",\n      \"identifier_type\": \"URL\",\n      \"relation_type\": \"IsDescribedBy\"\n    },\n    {\n      \"identifier\": \"10.5281/zenodo.5831879\",\n      \"identifier_type\": \"DOI\",\n      \"relation_type\": \"IsPublishedIn\"\n    },\n    {\n      \"identifier\": \"https://nemar.org/dataset/nm000195\",\n      \"identifier_type\": \"URL\",\n      \"relation_type\": \"IsDescribedBy\"\n    }\n  ],\n  \"funding_references\": [\n    {\n      \"funder_name\": \"Special Research Fund of Ghent University\"\n    },\n    {\n      \"funder_name\": \"DFG\",\n      \"award_number\": \"EXC 1086\",\n      \"award_title\": \"BrainLinks-BrainTools Cluster of Excellence\"\n    },\n    {\n      \"funder_name\": \"DFG\",\n      \"award_number\": \"INST 39/963-1 FUGG\",\n      \"award_title\": \"bwHPC initiative\"\n    },\n    {\n      \"funder_name\": \"European Commission\",\n      \"award_number\": \"657679\",\n      \"award_title\": \"Marie Sklodowska-Curie grant agreement\"\n    },\n    {\n      \"funder_name\": \"Ghent University\",\n      \"award_title\": \"Special Research Fund\"\n    },\n    {\n      \"funder_name\": \"DFG\",\n      \"award_number\": \"SPP 1527, MU 987/14-1\"\n    },\n    {\n      \"funder_name\": \"BMBF\",\n      \"award_number\": \"2017-0-00451\"\n    },\n    {\n      \"funder_name\": \"IITP\",\n      \"award_number\": \"1IS14013A\",\n      \"award_title\": \"Brain Korea 21 Plus Program\"\n    }\n  ],\n  \"resource_type_general\": \"Dataset\",\n  \"resource_type_specific\": \"EEG Dataset\",\n  \"modalities\": [\n    \"eeg\"\n  ],\n  \"sizes\": [\n    \"5.2 GB (721 files)\"\n  ],\n  \"formats\": [\n    \".bdf\",\n    \".json\",\n    \".md\",\n    \".tsv\",\n    \".vhdr\",\n    \".yaml\",\n    \".yml\"\n  ],\n  \"source_hash\": \"d0635377c418c944440f701a5f39c9f274ad5a50dd994398cbe58a1d2b584d05\"\n}","last_activity_at":"2026-08-16 13:31:33","source":null,"source_id":null,"subject_count":12,"modalities":"eeg","age_min":26,"age_max":26,"file_size":5168748037,"total_files":2711,"tasks":"p300","metadata_columns_error":null,"staleness_warn_stage":null,"staleness_admin_notified_at":null,"authors":"David Hübner, Thibault Verhoeven, Klaus-Robert Müller, Pieter-Jan Kindermans, Michael Tangermann","license":"CC-BY-4.0","readme":"[![DOI](https://img.shields.io/badge/DOI-10.82901%2Fnemar.nm000195-blue)](https://doi.org/10.82901/nemar.nm000195)\n\n# Mixture of LLP and EM for a visual matrix speller (ERP) dataset from\n\nMixture of LLP and EM for a visual matrix speller (ERP) dataset from Hübner et al 2018 [1]_.\n\n## Dataset Overview\n\n- **Code**: Huebner2018\n- **Paradigm**: p300\n- **DOI**: 10.1109/MCI.2018.2807039\n- **Subjects**: 12\n- **Sessions per subject**: 3\n- **Events**: Target=10002, NonTarget=10001\n- **Trial interval**: [-0.2, 0.7] s\n- **Session IDs**: 0, 1, 2\n- **File format**: BrainVision\n\n## Acquisition\n\n- **Sampling rate**: 1000.0 Hz\n- **Number of channels**: 31\n- **Channel types**: eeg=31, misc=6\n- **Channel names**: C3, C4, CP1, CP2, CP5, CP6, Cz, EOGvu, F10, F3, F4, F7, F8, F9, FC1, FC2, FC5, FC6, Fp1, Fp2, Fz, O1, O2, P10, P3, P4, P7, P8, P9, Pz, T7, T8, x_EMGl, x_GSR, x_Optic, x_Pulse, x_Respi\n- **Montage**: extended 10-20\n- **Hardware**: BrainAmp DC\n- **Software**: BBCI toolbox\n- **Reference**: nose\n- **Sensor type**: Ag/AgCl\n- **Line frequency**: 50.0 Hz\n- **Impedance threshold**: 20.0 kOhm\n- **Cap manufacturer**: EasyCap\n\n## Participants\n\n- **Number of subjects**: 12\n- **Health status**: healthy\n- **Age**: mean=26, min=19, max=31\n- **Gender distribution**: female=8, male=4\n- **BCI experience**: mixed\n- **Species**: human\n\n## Experimental Protocol\n\n- **Paradigm**: p300\n- **Number of classes**: 2\n- **Class labels**: Target, NonTarget\n- **Trial duration**: 17.0 s\n- **Tasks**: copy-spelling\n- **Study design**: Visual ERP copy-spelling task using a modified 6x6 grid extended with 10 # symbols as visual blanks, using flexible highlighting scheme with two interleaved sequences to enable unsupervised learning methods (EM, LLP, MIX)\n- **Feedback type**: visual\n- **Stimulus type**: modified matrix speller with flexible highlighting\n- **Stimulus modalities**: visual\n- **Primary modality**: visual\n- **Mode**: online\n- **Instructions**: copy-spelling task - spell German sentence 'Franzy jagt im Taxi quer durch das'\n- **Stimulus presentation**: soa_ms=250, stimulus_duration_ms=100, isi_ms=150, highlighting_type=combination of brightness enhancement, rotation, enlargement and trichromatic grid overlay, distance_to_screen_cm=80, screen_size_inches=24\n\n## HED Event Annotations\n\nSchema: HED 8.4.0 | Browse: https://www.hedtags.org/hed-schema-browser\n\n```\n  Target\n    ├─ Sensory-event\n    ├─ Experimental-stimulus\n    ├─ Visual-presentation\n    └─ Target\n\n  NonTarget\n    ├─ Sensory-event\n    ├─ Experimental-stimulus\n    ├─ Visual-presentation\n    └─ Non-target\n\n```\n## Paradigm-Specific Parameters\n\n- **Detected paradigm**: p300\n- **Number of targets**: 46\n- **Inter-stimulus interval**: 150.0 ms\n- **Stimulus onset asynchrony**: 250.0 ms\n\n## Data Structure\n\n- **Trials**: 35\n- **Blocks per session**: 3\n- **Trials context**: 35 characters per block (one trial = spelling one character), 3 blocks per session (one block per unsupervised algorithm: EM, LLP, MIX in pseudo-randomized order)\n\n## Preprocessing\n\n- **Data state**: raw\n- **Preprocessing applied**: False\n\n## Signal Processing\n\n- **Classifiers**: EM (Expectation-Maximization), LLP (Learning from Label Proportions), MIX (mixture of EM and LLP), shrinkage-regularized LDA (Ledoit-Wolf), Bayesian least square regression\n- **Feature extraction**: mean amplitudes in six temporal intervals per channel\n\n## Cross-Validation\n\n- **Method**: leave-one-character-out for offline analysis; online sequential testing\n- **Evaluation type**: online, within_session, unsupervised_learning\n\n## Performance (Original Study)\n\n- **Accuracy**: 80.0%\n- **Mix Auc After 7 Chars**: 80.0\n- **Time To 80 Accuracy Seconds**: 168.0\n- **Epochs To 80 Accuracy**: 476.0\n- **Characters To 80 Accuracy**: 7.0\n\n## BCI Application\n\n- **Applications**: speller, communication\n- **Environment**: controlled laboratory\n- **Online feedback**: True\n\n## Tags\n\n- **Pathology**: Healthy\n- **Modality**: Visual\n- **Type**: Research\n\n## Documentation\n\n- **DOI**: 10.5281/zenodo.192684\n- **Associated paper DOI**: 10.1109/MCI.2018.2807039\n- **License**: CC-BY-4.0\n- **Investigators**: David Hübner, Thibault Verhoeven, Klaus-Robert Müller, Pieter-Jan Kindermans, Michael Tangermann\n- **Contact**: p.kindermans@tu-berlin.de; michael.tangermann@blbt.uni-freiburg.de\n- **Institution**: University of Freiburg\n- **Department**: Brain State Decoding Lab\n- **Address**: Brain State Decoding Lab, University of Freiburg, Freiburg, GERMANY\n- **Country**: DE\n- **Repository**: Zenodo\n- **Data URL**: https://zenodo.org/record/5831879\n- **Publication year**: 2018\n- **Funding**: BrainLinks-BrainTools Cluster of Excellence funded by the German Research Foundation (DFG), grant number EXC 1086; bwHPC initiative, grant INST 39/963-1 FUGG; European Union's Horizon 2020 research and innovation program under the Marie Sklodowska-Curie grant agreement NO 657679; Special Research Fund of Ghent University; DFG (DFG SPP 1527, MU 987/14-1); Federal Ministry for Education and Research (BMBF No. 2017-0-00451); Brain Korea 21 Plus Program by the Institute for Information & Communications Technology Promotion (IITP) grant (1IS14013A) funded by the Korean government\n- **Ethics approval**: University Medical Center Freiburg ethics committee\n- **Keywords**: unsupervised learning, brain-computer interface, event-related potentials, P300 speller, expectation-maximization, learning from label proportions, MIX method, EEG\n\n## Abstract\n\nOne of the fundamental challenges in brain-computer interfaces (BCIs) is to tune a brain signal decoder to reliably detect a user's intention. While information about the decoder can partially be transferred between subjects or sessions, optimal decoding performance can only be reached with novel data from the current session. Thus, it is preferable to learn from unlabeled data gained from the actual usage of the BCI application instead of conducting a calibration recording prior to BCI usage. We review such unsupervised machine learning methods for BCIs based on event-related potentials of the electroencephalogram. We present results of an online study with twelve healthy participants controlling a visual speller. Online performance is reported for three completely unsupervised learning methods: (1) learning from label proportions, (2) an expectation-maximization approach and (3) MIX, which combines the strengths of the two other methods. After a short ramp-up, we observed that the MIX method not only defeats its two unsupervised competitors but even performs on par with a state-of-the-art regularized linear discriminant analysis trained on the same number of data points and with full label access. With this online study, we deliver the best possible proof in BCI that an unsupervised decoding method can in practice render a supervised method unnecessary. This is possible despite skipping the calibration, without losing much performance and with the prospect of continuous improvement over a session. Thus, our findings pave the way for a transition from supervised to unsupervised learning methods in BCIs based on event-related potentials.\n\n## Methodology\n\nOnline study comparing three unsupervised learning methods (EM, LLP, MIX) for P300 speller. Twelve healthy volunteers (8 female, 4 male, mean age 26, range 19-31 years) participated in a single session each. Subjects spelled the German sentence 'Franzy jagt im Taxi quer durch das' (35 characters) in three blocks, each using a different unsupervised algorithm in pseudo-randomized order. Each trial (spelling one character) consisted of 68 highlighting events with 250 ms SOA and 100 ms stimulus duration (ISI=150 ms). The speller used a modified 6x6 grid with 36 normal characters extended with 10 # symbols as visual blanks (total 46 symbols). Two interleaved highlighting sequences were used: S1 highlighted only normal characters, S2 highlighted both normal characters and # symbols, creating different known target-to-non-target ratios to enable learning from label proportions. Highlighting consisted of brightness enhancement, rotation, enlargement and trichromatic grid overlay. Classifiers were randomly initialized at block start and updated after ea","bids_version":"1.9.0","sessions_count":3,"publish_date":"2026-06-02 01:27:42","embedding_dirty":0,"license_tier":"attribution","zarr_status":"ready","zarr_converted_at":"2026-09-04 11:10:11","zarr_store_count":360,"zarr_index_etag":"424ed8a097d73678e7a684dddc59c693","zarr_source_commit":"6d0d83466e5b45625c598cfa59567a24b69b6c0d","archive_status":"ready","archive_size":3962040174,"archive_retry_count":0,"records_status":"ready","archive_skip_reason":null,"zarr_errors":0,"zarr_failure_count":0,"zarr_deterministic":0,"zarr_failed_at":null,"num_dataset_citations":0,"num_datapaper_citations":26,"n_channels":31,"electrode_system":"10-10","has_hed":1,"hed_version":"8.4.0","is_exemplar":0,"bytes_present":5159341135,"data_complete":1,"withdrawn_at":null,"withdrawn_reason":null,"archive_complete":1,"archive_absent_files":0,"archive_declared_files":2711,"zarr_pool_breaks":0,"total_recording_duration":55334,"recording_duration_min":126,"recording_duration_max":183,"recording_count":360,"recordings_unavailable":0,"recordings_measured":360,"channel_count_min":31,"channel_count_max":31,"sampling_frequency":1000,"power_line_frequency":50,"eeg_reference":"nose","placement_scheme":"extended 10-20","sweep_stamps":"{\"enrichment_updated_at\":\"2026-08-18 18:12:16\",\"metadata_updated_at\":\"2026-08-18 21:09:52\",\"archive_checked_at\":\"2026-08-18 21:13:23\",\"zarr_checked_at\":\"2026-06-07 17:58:28\",\"records_checked_at\":\"2026-08-18 21:10:55\",\"citations_updated_at\":\"2026-09-08 03:00:48\",\"channel_montage_checked_at\":\"2026-06-28 22:57:14\",\"hed_checked_at\":\"2026-06-30 07:27:26\",\"data_checked_at\":null,\"availability_report_at\":\"2026-08-20 03:01:18\",\"signal_defaults_at\":\"2026-09-02 11:43:07\",\"recording_stats_at\":\"2026-09-05 03:01:12\"}","participants":12,"num_citations":26,"latest_version":"v1.0.2","zarr_verify_status":null,"zarr_verified_at":null,"owner_username":"bruaristimunha","owner_github":"bruAristimunha","file_size_formatted":"4.81 GB","zarr_data_failures":null,"zarr_index_url":"https://zarr.nemar.org/nm000195/zarr/index.json","attestation_deposit_type":null,"attestation_key_status":null,"attestation_deidentified":null,"attestation_no_duplicate":null,"attestation_upstream_source":null,"attestation_accepted_at":null}}