<?xml version="1.0" encoding="UTF-8"?>
<rdf:RDF xmlns="http://purl.org/rss/1.0/"
 xmlns:dc="http://purl.org/dc/elements/1.1/"
 xmlns:dcterms="http://purl.org/dc/terms/"
 xmlns:cc="http://web.resource.org/cc/"
 xmlns:prism="http://prismstandard.org/namespaces/basic/2.0/"
 xmlns:rdf="http://www.w3.org/1999/02/22-rdf-syntax-ns#"
 xmlns:admin="http://webns.net/mvcb/"
 xmlns:content="http://purl.org/rss/1.0/modules/content/">
    <channel rdf:about="https://www.mdpi.com/rss/journal/BDCC">
		<title>Big Data and Cognitive Computing</title>
		<description>Latest open access articles published in Big Data Cogn. Comput. at https://www.mdpi.com/journal/BDCC</description>
		<link>https://www.mdpi.com/journal/BDCC</link>
		<admin:generatorAgent rdf:resource="https://www.mdpi.com/journal/BDCC"/>
		<admin:errorReportsTo rdf:resource="mailto:support@mdpi.com"/>
		<dc:publisher>MDPI</dc:publisher>
		<dc:language>en</dc:language>
		<dc:rights>Creative Commons Attribution (CC-BY)</dc:rights>
						<prism:copyright>MDPI</prism:copyright>
		<prism:rightsAgent>support@mdpi.com</prism:rightsAgent>
		<image rdf:resource="https://pub.mdpi-res.com/img/design/mdpi-pub-logo.png?13cf3b5bd783e021?1788181901"/>
				<items>
			<rdf:Seq>
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/9/295" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/9/294" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/9/293" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/9/292" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/9/291" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/9/290" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/9/289" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/9/288" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/9/287" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/9/286" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/9/285" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/9/284" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/9/283" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/8/282" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/8/281" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/8/280" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/8/279" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/8/278" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/8/277" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/8/276" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/8/275" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/8/274" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/8/273" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/8/272" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/8/271" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/8/270" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/8/269" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/8/268" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/8/267" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/8/266" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/8/265" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/8/264" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/8/263" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/8/262" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/8/261" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/8/260" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/8/259" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/8/258" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/8/257" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/8/256" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/8/255" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/8/254" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/8/253" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/8/252" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/8/251" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/8/250" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/8/249" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/8/248" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/8/247" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/8/246" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/8/245" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/244" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/243" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/242" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/241" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/240" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/239" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/238" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/237" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/236" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/235" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/234" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/233" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/232" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/231" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/230" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/229" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/228" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/227" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/226" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/225" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/224" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/223" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/222" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/221" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/220" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/219" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/218" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/217" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/216" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/215" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/214" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/213" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/212" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/211" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/210" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/209" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/208" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/207" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/206" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/205" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/204" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/203" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/202" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/201" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/200" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/7/199" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/6/198" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/6/197" />
            				<rdf:li rdf:resource="https://www.mdpi.com/2504-2289/10/6/195" />
                    	</rdf:Seq>
		</items>
				<cc:license rdf:resource="https://creativecommons.org/licenses/by/4.0/" />
	</channel>

        <item rdf:about="https://www.mdpi.com/2504-2289/10/9/295">

	<title>BDCC, Vol. 10, Pages 295: Proof-Carrying Neuro-Symbolic Reasoning for Non-Monotonic Legal Decision Support with LLMs</title>
	<link>https://www.mdpi.com/2504-2289/10/9/295</link>
	<description>Large language models (LLMs) and retrieval-augmented generation (RAG) are increasingly used in legal decision support, but retrieved evidence and fluent explanations do not guarantee valid normative inference. This paper proposes a proof-carrying neuro-symbolic method for non-monotonic legal reasoning. The LLM component is restricted to source-linked extraction of facts, defeasible rules, defeaters, priorities, citations, and operational confidence scores, while a deterministic symbolic engine computes the conclusion. Evidence is represented as a finite defeasible normative theory and compiled into a Dung-style argumentation framework; accepted conclusions are obtained from the grounded extension and returned with proof graphs showing support, attacks, and priority-based defeats. Under gold formalization, the symbolic engine achieved 99.3% accuracy on a 600-case controlled benchmark. In a 240-scenario LLM-to-logic experiment, the GPT-4o extractor followed by symbolic reasoning achieved 86.7% downstream accuracy versus 75.8% for a direct LLM over the same retrieved evidence; the paired difference was supported by an exact McNemar test after Holm correction (adjusted p = 0.016). Differences from the PDL and simpler symbolic baselines were not statistically established. Validation-triggered repair yielded 90.4% observed accuracy. Public-contract, Russian-law, stress-test, scalability, and lawyer-verification experiments further delimit the feasibility and current limitations of proof-carrying legal decision support.</description>
	<pubDate>2026-09-01</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 295: Proof-Carrying Neuro-Symbolic Reasoning for Non-Monotonic Legal Decision Support with LLMs</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/9/295">doi: 10.3390/bdcc10090295</a></p>
	<p>Authors:
		Maxim Ulizko
		Tatiana Polevaya
		Ivan Tomilov
		Natalia Gusarova
		Aleksandra Vatian
		</p>
	<p>Large language models (LLMs) and retrieval-augmented generation (RAG) are increasingly used in legal decision support, but retrieved evidence and fluent explanations do not guarantee valid normative inference. This paper proposes a proof-carrying neuro-symbolic method for non-monotonic legal reasoning. The LLM component is restricted to source-linked extraction of facts, defeasible rules, defeaters, priorities, citations, and operational confidence scores, while a deterministic symbolic engine computes the conclusion. Evidence is represented as a finite defeasible normative theory and compiled into a Dung-style argumentation framework; accepted conclusions are obtained from the grounded extension and returned with proof graphs showing support, attacks, and priority-based defeats. Under gold formalization, the symbolic engine achieved 99.3% accuracy on a 600-case controlled benchmark. In a 240-scenario LLM-to-logic experiment, the GPT-4o extractor followed by symbolic reasoning achieved 86.7% downstream accuracy versus 75.8% for a direct LLM over the same retrieved evidence; the paired difference was supported by an exact McNemar test after Holm correction (adjusted p = 0.016). Differences from the PDL and simpler symbolic baselines were not statistically established. Validation-triggered repair yielded 90.4% observed accuracy. Public-contract, Russian-law, stress-test, scalability, and lawyer-verification experiments further delimit the feasibility and current limitations of proof-carrying legal decision support.</p>
	]]></content:encoded>

	<dc:title>Proof-Carrying Neuro-Symbolic Reasoning for Non-Monotonic Legal Decision Support with LLMs</dc:title>
			<dc:creator>Maxim Ulizko</dc:creator>
			<dc:creator>Tatiana Polevaya</dc:creator>
			<dc:creator>Ivan Tomilov</dc:creator>
			<dc:creator>Natalia Gusarova</dc:creator>
			<dc:creator>Aleksandra Vatian</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10090295</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-09-01</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-09-01</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>9</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>295</prism:startingPage>
		<prism:doi>10.3390/bdcc10090295</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/9/295</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/9/294">

	<title>BDCC, Vol. 10, Pages 294: Isolating Graph Topology from Model Architecture in GNN-Based Fraud Detection: An Empirical Framework</title>
	<link>https://www.mdpi.com/2504-2289/10/9/294</link>
	<description>Graph topology and model architecture are routinely co-designed in GNN-based fraud detection, making it impossible to attribute performance gains to either component. We address this by fixing the training loop, features, and evaluation protocol while independently varying the graph construction strategy and GNN architecture. Three strategies are evaluated: multi-relation temporal, hybrid structural similarity, and intra-group, each evaluated across three GNN architectures (GATv2, GCN, and GraphSAGE). A feature-identical MLP with no graph structure serves as an empirical anchor. On the Sparkov dataset, all three GNN strategies exceed the MLP by 5.0&amp;amp;ndash;9.7 F1 points, confirming that topology contributes genuine discriminative value when per-cardholder histories are dense. On the IBM dataset, the intra-group strategy collapses 6.9 F1 points below the MLP, a consequence of near-empty neighbourhoods (mean degree 1.1) following down-sampling, while multi-relation retains ranking advantages in AUC (Area Under the Curve; 0.985) and Average Precision (0.908) despite marginal F1 parity. Across both datasets, multi-relation temporal construction is the highest-performing strategy, achieving F1 0.908 and AUC 0.994 on Sparkov and F1 0.831 on IBM, though this ranking is not entirely architecture-independent: GraphSAGE narrowly inverts the multi-relation/intra-group order on Sparkov. These results suggest that graph construction quality, rather than mere graph presence, matters more for GNN performance than connectivity alone under the conditions evaluated here.</description>
	<pubDate>2026-09-01</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 294: Isolating Graph Topology from Model Architecture in GNN-Based Fraud Detection: An Empirical Framework</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/9/294">doi: 10.3390/bdcc10090294</a></p>
	<p>Authors:
		Roya Amiri
		Sardar Jaf
		</p>
	<p>Graph topology and model architecture are routinely co-designed in GNN-based fraud detection, making it impossible to attribute performance gains to either component. We address this by fixing the training loop, features, and evaluation protocol while independently varying the graph construction strategy and GNN architecture. Three strategies are evaluated: multi-relation temporal, hybrid structural similarity, and intra-group, each evaluated across three GNN architectures (GATv2, GCN, and GraphSAGE). A feature-identical MLP with no graph structure serves as an empirical anchor. On the Sparkov dataset, all three GNN strategies exceed the MLP by 5.0&amp;amp;ndash;9.7 F1 points, confirming that topology contributes genuine discriminative value when per-cardholder histories are dense. On the IBM dataset, the intra-group strategy collapses 6.9 F1 points below the MLP, a consequence of near-empty neighbourhoods (mean degree 1.1) following down-sampling, while multi-relation retains ranking advantages in AUC (Area Under the Curve; 0.985) and Average Precision (0.908) despite marginal F1 parity. Across both datasets, multi-relation temporal construction is the highest-performing strategy, achieving F1 0.908 and AUC 0.994 on Sparkov and F1 0.831 on IBM, though this ranking is not entirely architecture-independent: GraphSAGE narrowly inverts the multi-relation/intra-group order on Sparkov. These results suggest that graph construction quality, rather than mere graph presence, matters more for GNN performance than connectivity alone under the conditions evaluated here.</p>
	]]></content:encoded>

	<dc:title>Isolating Graph Topology from Model Architecture in GNN-Based Fraud Detection: An Empirical Framework</dc:title>
			<dc:creator>Roya Amiri</dc:creator>
			<dc:creator>Sardar Jaf</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10090294</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-09-01</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-09-01</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>9</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>294</prism:startingPage>
		<prism:doi>10.3390/bdcc10090294</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/9/294</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/9/293">

	<title>BDCC, Vol. 10, Pages 293: CapAgent: Semantic Data-Flow Governance for LLM Agents in Big-Data Cognitive Computing</title>
	<link>https://www.mdpi.com/2504-2289/10/9/293</link>
	<description>Large language model (LLM) agents are becoming cognitive interfaces to data lakes, enterprise knowledge bases, vector memories, browsers, files, and software tools. This shift creates a data-governance gap: an agent may reason over large private context, yet the protected resource often lacks a verifiable record of which user intent, data object, action, destination, and semantic release were authorized. This paper proposes CapAgent, a semantic data-flow governance middleware for LLM agents in big-data cognitive-computing environments. CapAgent maps human-attested task intent into signed, attenuable, and purpose-bound capability tokens that are checked by a reference monitor before sensitive tool invocation, memory retrieval, data export, and inter-agent delegation. Its policy layer combines task templates, resource labels, destination rules, caveats, semantic release modes, and audit obligations; its runtime enforces both symbolic scope checks and semantic recoverability checks over protected facts. We present formal governance semantics, a conservative intent compiler, an explainable data-flow decision workflow, and a runnable Python middleware. A reproducible trace-replay benchmark with 600 benign and adversarial traces across five data-intensive agent scenarios reports attack success, benign success, false blocking, latency, component ablations, and audit quality. In this synthetic trace-replay evaluation, the full monitor reduces measured attack success from 100.00% under ambient execution and 12.50% under scope-only authorization to 0.00% (Wilson 95% CI [0.00, 0.95]), while retaining 75.00% benign success. In addition, we conduct a 520-trial end-to-end tool-calling benchmark with representative prompt-only, task-shield-style, CaMeL-style, scope-only, and full-CapAgent configurations; a 210-task compiler gold-standard evaluation; a 240-item semantic-release calibration set; and a 12-cell BDCC-style scalability microbenchmark. In these supplemental tests, full CapAgent obtains 0.00% ASR (95% CI [0.00, 1.06]) in the tool-calling benchmark, 87.50% exact-policy compiler match with 0.00% over-authorization, 88.89% semantic-release recall with 0.00% false-block rate, and sub-millisecond in-process authorization latency up to 100,000 resources. The results support CapAgent as an auditable governance layer for cognitive LLM agents rather than as a replacement for model-level alignment or public end-to-end agent benchmarks.</description>
	<pubDate>2026-09-01</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 293: CapAgent: Semantic Data-Flow Governance for LLM Agents in Big-Data Cognitive Computing</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/9/293">doi: 10.3390/bdcc10090293</a></p>
	<p>Authors:
		Huiying Hou
		Yucong Ma
		Jianyu Miao
		</p>
	<p>Large language model (LLM) agents are becoming cognitive interfaces to data lakes, enterprise knowledge bases, vector memories, browsers, files, and software tools. This shift creates a data-governance gap: an agent may reason over large private context, yet the protected resource often lacks a verifiable record of which user intent, data object, action, destination, and semantic release were authorized. This paper proposes CapAgent, a semantic data-flow governance middleware for LLM agents in big-data cognitive-computing environments. CapAgent maps human-attested task intent into signed, attenuable, and purpose-bound capability tokens that are checked by a reference monitor before sensitive tool invocation, memory retrieval, data export, and inter-agent delegation. Its policy layer combines task templates, resource labels, destination rules, caveats, semantic release modes, and audit obligations; its runtime enforces both symbolic scope checks and semantic recoverability checks over protected facts. We present formal governance semantics, a conservative intent compiler, an explainable data-flow decision workflow, and a runnable Python middleware. A reproducible trace-replay benchmark with 600 benign and adversarial traces across five data-intensive agent scenarios reports attack success, benign success, false blocking, latency, component ablations, and audit quality. In this synthetic trace-replay evaluation, the full monitor reduces measured attack success from 100.00% under ambient execution and 12.50% under scope-only authorization to 0.00% (Wilson 95% CI [0.00, 0.95]), while retaining 75.00% benign success. In addition, we conduct a 520-trial end-to-end tool-calling benchmark with representative prompt-only, task-shield-style, CaMeL-style, scope-only, and full-CapAgent configurations; a 210-task compiler gold-standard evaluation; a 240-item semantic-release calibration set; and a 12-cell BDCC-style scalability microbenchmark. In these supplemental tests, full CapAgent obtains 0.00% ASR (95% CI [0.00, 1.06]) in the tool-calling benchmark, 87.50% exact-policy compiler match with 0.00% over-authorization, 88.89% semantic-release recall with 0.00% false-block rate, and sub-millisecond in-process authorization latency up to 100,000 resources. The results support CapAgent as an auditable governance layer for cognitive LLM agents rather than as a replacement for model-level alignment or public end-to-end agent benchmarks.</p>
	]]></content:encoded>

	<dc:title>CapAgent: Semantic Data-Flow Governance for LLM Agents in Big-Data Cognitive Computing</dc:title>
			<dc:creator>Huiying Hou</dc:creator>
			<dc:creator>Yucong Ma</dc:creator>
			<dc:creator>Jianyu Miao</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10090293</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-09-01</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-09-01</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>9</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>293</prism:startingPage>
		<prism:doi>10.3390/bdcc10090293</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/9/293</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/9/292">

	<title>BDCC, Vol. 10, Pages 292: PASM-Net: A Coordinate-Aware Manifold-Regularized Framework for Open-Set UAV RF Fingerprint Identification</title>
	<link>https://www.mdpi.com/2504-2289/10/9/292</link>
	<description>To address the challenges of classifying known categories and rejecting unknown categories in the radio-frequency fingerprinting of uncooperative UAVs in open, low-altitude environments, this paper proposes PASM-Net, an open-set recognition method based on coordinate awareness and manifold continuity regularization. Existing deep learning-based RFFI methods still suffer from two shortcomings in open scenarios: First, while conventional convolutions and global pooling yield compact representations, they may weaken position-related information in the time&amp;amp;ndash;frequency spectrum, thereby affecting the model&amp;amp;rsquo;s ability to characterize frequency-hopping trajectories and local time&amp;amp;ndash;frequency structures; on the other hand, decision mechanisms that rely solely on Softmax classifiers or simple distance metrics are susceptible to signal amplitude fluctuations and intra-class distribution dispersion, leading to the misclassification of unknown samples into known classes. To address these issues, this paper designs the PASM-Net feature-learning framework. First, the model employs multiscale hollow convolutions to extract local time&amp;amp;ndash;frequency features across different receptive fields and introduces a coordinate attention mechanism (Coordinate Attention, CoordAtt) prior to global pooling, thereby enhancing the model&amp;amp;rsquo;s ability to represent position-related information in both the time and frequency domains. Second, the model introduces manifold continuity regularization (MCR) in the fused semantic space. By constraining feature variations within local neighborhoods via a Laplacian regularization term, MCR reduces the dispersion of known-class feature distributions. Finally, the model employs ArcFace to enhance the angular separability among known classes and performs open-set classification based on the cosine distance between test samples and the centers of known classes. The experimental results across six open-set scenarios on the DroneRFb-Spectra dataset showed that PASM-Net achieved an average true unknown rate (TUR) of 99.48%, an unknown accuracy of 97.9%, and an unknown-class precision (UP) of 87.3%, while maintaining a high performance in known-class recognition. Under the current experimental setup, this effectively improves the trade-off between known-class classification and unknown-class rejection.</description>
	<pubDate>2026-08-31</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 292: PASM-Net: A Coordinate-Aware Manifold-Regularized Framework for Open-Set UAV RF Fingerprint Identification</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/9/292">doi: 10.3390/bdcc10090292</a></p>
	<p>Authors:
		Mingjun Jiang
		Zhongqiang Luo
		Jia Yuan
		</p>
	<p>To address the challenges of classifying known categories and rejecting unknown categories in the radio-frequency fingerprinting of uncooperative UAVs in open, low-altitude environments, this paper proposes PASM-Net, an open-set recognition method based on coordinate awareness and manifold continuity regularization. Existing deep learning-based RFFI methods still suffer from two shortcomings in open scenarios: First, while conventional convolutions and global pooling yield compact representations, they may weaken position-related information in the time&amp;amp;ndash;frequency spectrum, thereby affecting the model&amp;amp;rsquo;s ability to characterize frequency-hopping trajectories and local time&amp;amp;ndash;frequency structures; on the other hand, decision mechanisms that rely solely on Softmax classifiers or simple distance metrics are susceptible to signal amplitude fluctuations and intra-class distribution dispersion, leading to the misclassification of unknown samples into known classes. To address these issues, this paper designs the PASM-Net feature-learning framework. First, the model employs multiscale hollow convolutions to extract local time&amp;amp;ndash;frequency features across different receptive fields and introduces a coordinate attention mechanism (Coordinate Attention, CoordAtt) prior to global pooling, thereby enhancing the model&amp;amp;rsquo;s ability to represent position-related information in both the time and frequency domains. Second, the model introduces manifold continuity regularization (MCR) in the fused semantic space. By constraining feature variations within local neighborhoods via a Laplacian regularization term, MCR reduces the dispersion of known-class feature distributions. Finally, the model employs ArcFace to enhance the angular separability among known classes and performs open-set classification based on the cosine distance between test samples and the centers of known classes. The experimental results across six open-set scenarios on the DroneRFb-Spectra dataset showed that PASM-Net achieved an average true unknown rate (TUR) of 99.48%, an unknown accuracy of 97.9%, and an unknown-class precision (UP) of 87.3%, while maintaining a high performance in known-class recognition. Under the current experimental setup, this effectively improves the trade-off between known-class classification and unknown-class rejection.</p>
	]]></content:encoded>

	<dc:title>PASM-Net: A Coordinate-Aware Manifold-Regularized Framework for Open-Set UAV RF Fingerprint Identification</dc:title>
			<dc:creator>Mingjun Jiang</dc:creator>
			<dc:creator>Zhongqiang Luo</dc:creator>
			<dc:creator>Jia Yuan</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10090292</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-08-31</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-08-31</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>9</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>292</prism:startingPage>
		<prism:doi>10.3390/bdcc10090292</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/9/292</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/9/291">

	<title>BDCC, Vol. 10, Pages 291: A Study on a Multimodal Agentic RAG Framework for Integrating Multi-Source Heterogeneous Information in Higher Education Institutions</title>
	<link>https://www.mdpi.com/2504-2289/10/9/291</link>
	<description>Against the backdrop of digital transformation in higher education, campus information is often dispersed across independent platforms, creating fragmented access and retrieval barriers. To address these challenges, this study proposes a multimodal agentic RAG framework for integrating multi-source, heterogeneous university information through a unified natural language interaction portal. Compared with basic RAG architectures, the framework introduces three improvements: (1) vision&amp;amp;ndash;language models convert unstructured visual content into indexable textual evidence; (2) BM25 sparse retrieval, dense-vector retrieval, and real-time online retrieval jointly support keyword matching, semantic retrieval, and up-to-date information access; and (3) agent-based dynamic routing adaptively schedules retrieval tools according to query intent and evidence sufficiency, reducing redundant calls. Experiments show that the proposed framework achieves higher overall performance in answer accuracy, factual consistency, evidence coverage, and composite score than general-purpose large language models, traditional single-path RAG, and representative advanced RAG methods including Self-RAG, Adaptive-RAG, and VisRAG. Ablation studies further validate the contributions of the core modules. Overall, the framework provides a low-disruption solution for unified information retrieval and natural language interaction in higher education scenarios.</description>
	<pubDate>2026-08-27</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 291: A Study on a Multimodal Agentic RAG Framework for Integrating Multi-Source Heterogeneous Information in Higher Education Institutions</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/9/291">doi: 10.3390/bdcc10090291</a></p>
	<p>Authors:
		Minghan Li
		Haotian Wang
		Zekai Sun
		Hang Zhou
		</p>
	<p>Against the backdrop of digital transformation in higher education, campus information is often dispersed across independent platforms, creating fragmented access and retrieval barriers. To address these challenges, this study proposes a multimodal agentic RAG framework for integrating multi-source, heterogeneous university information through a unified natural language interaction portal. Compared with basic RAG architectures, the framework introduces three improvements: (1) vision&amp;amp;ndash;language models convert unstructured visual content into indexable textual evidence; (2) BM25 sparse retrieval, dense-vector retrieval, and real-time online retrieval jointly support keyword matching, semantic retrieval, and up-to-date information access; and (3) agent-based dynamic routing adaptively schedules retrieval tools according to query intent and evidence sufficiency, reducing redundant calls. Experiments show that the proposed framework achieves higher overall performance in answer accuracy, factual consistency, evidence coverage, and composite score than general-purpose large language models, traditional single-path RAG, and representative advanced RAG methods including Self-RAG, Adaptive-RAG, and VisRAG. Ablation studies further validate the contributions of the core modules. Overall, the framework provides a low-disruption solution for unified information retrieval and natural language interaction in higher education scenarios.</p>
	]]></content:encoded>

	<dc:title>A Study on a Multimodal Agentic RAG Framework for Integrating Multi-Source Heterogeneous Information in Higher Education Institutions</dc:title>
			<dc:creator>Minghan Li</dc:creator>
			<dc:creator>Haotian Wang</dc:creator>
			<dc:creator>Zekai Sun</dc:creator>
			<dc:creator>Hang Zhou</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10090291</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-08-27</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-08-27</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>9</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>291</prism:startingPage>
		<prism:doi>10.3390/bdcc10090291</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/9/291</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/9/290">

	<title>BDCC, Vol. 10, Pages 290: Cross-Cultural Scenario Benchmark: Evaluating LLMs&amp;rsquo; Cross-Cultural Understanding</title>
	<link>https://www.mdpi.com/2504-2289/10/9/290</link>
	<description>Cross-cultural reasoning and alignment have been identified as key weaknesses of large language models (LLMs), but the architectural or cognitive features underlying these failures have not been adequately examined. In addition, previous studies rely almost exclusively on datasets and benchmarks constructed under the WEIRD (Western, Educated, Industrialized, Rich, and Democratic) bias. To address this data bias issue, we prepare a dataset with substantial coverage of non-WEIRD cultures and five-dimensional (W, E, I, R, and D) annotations. This dataset supports an interpretable approach to examining weaknesses in LLMs&amp;amp;rsquo; cross-cultural alignment. We adopt the Chinese&amp;amp;ndash;Foreign Cultural Differences Case Repository at Xiamen University, which contains 9342 cases across 151 countries, 6 continents, and 10 cultural domains. These cases are processed and transformed into benchmark-ready structured data through topic normalization, structured metadata cleaning, continent correction, and country-level WEIRD annotation along five dimensions. Each case is converted into a six-option cultural attribution question with five cognitive-trap distractors grounded in cognitive reasoning and pragmatic interpretation. Evaluation of 6 mainstream large language models shows that their dominant failures do not involve explicit stereotypes. Instead, 61% of all errors arise from oversimplifying complex cultural phenomena or applying familiar cultural frames. The proportion of errors that explain specific cultural conflicts through seemingly universal value frames increases from 11% at the low-WEIRD end to 20% at the high-WEIRD end of the dataset. These results suggest that WEIRD data bias reflects both the underrepresentation of low-WEIRD cultures and the overactivation of dominant value frames in high-WEIRD contexts.</description>
	<pubDate>2026-08-27</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 290: Cross-Cultural Scenario Benchmark: Evaluating LLMs&amp;rsquo; Cross-Cultural Understanding</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/9/290">doi: 10.3390/bdcc10090290</a></p>
	<p>Authors:
		Mengxi Guo
		Leiming Gao
		Winnie Zeng
		Gaoqi Rao
		Chu-Ren Huang
		</p>
	<p>Cross-cultural reasoning and alignment have been identified as key weaknesses of large language models (LLMs), but the architectural or cognitive features underlying these failures have not been adequately examined. In addition, previous studies rely almost exclusively on datasets and benchmarks constructed under the WEIRD (Western, Educated, Industrialized, Rich, and Democratic) bias. To address this data bias issue, we prepare a dataset with substantial coverage of non-WEIRD cultures and five-dimensional (W, E, I, R, and D) annotations. This dataset supports an interpretable approach to examining weaknesses in LLMs&amp;amp;rsquo; cross-cultural alignment. We adopt the Chinese&amp;amp;ndash;Foreign Cultural Differences Case Repository at Xiamen University, which contains 9342 cases across 151 countries, 6 continents, and 10 cultural domains. These cases are processed and transformed into benchmark-ready structured data through topic normalization, structured metadata cleaning, continent correction, and country-level WEIRD annotation along five dimensions. Each case is converted into a six-option cultural attribution question with five cognitive-trap distractors grounded in cognitive reasoning and pragmatic interpretation. Evaluation of 6 mainstream large language models shows that their dominant failures do not involve explicit stereotypes. Instead, 61% of all errors arise from oversimplifying complex cultural phenomena or applying familiar cultural frames. The proportion of errors that explain specific cultural conflicts through seemingly universal value frames increases from 11% at the low-WEIRD end to 20% at the high-WEIRD end of the dataset. These results suggest that WEIRD data bias reflects both the underrepresentation of low-WEIRD cultures and the overactivation of dominant value frames in high-WEIRD contexts.</p>
	]]></content:encoded>

	<dc:title>Cross-Cultural Scenario Benchmark: Evaluating LLMs&amp;amp;rsquo; Cross-Cultural Understanding</dc:title>
			<dc:creator>Mengxi Guo</dc:creator>
			<dc:creator>Leiming Gao</dc:creator>
			<dc:creator>Winnie Zeng</dc:creator>
			<dc:creator>Gaoqi Rao</dc:creator>
			<dc:creator>Chu-Ren Huang</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10090290</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-08-27</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-08-27</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>9</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>290</prism:startingPage>
		<prism:doi>10.3390/bdcc10090290</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/9/290</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/9/289">

	<title>BDCC, Vol. 10, Pages 289: Learning Multi-View Representations with Graph-Level Fusion: A Bi-Level Optimization Perspective</title>
	<link>https://www.mdpi.com/2504-2289/10/9/289</link>
	<description>Multi-view representation learning, aiming to extract discriminative patterns from heterogeneous data sources, has attracted increasing attention in recent years. A central challenge in multi-view data fusion lies in determining the appropriate weight and importance of each view in a principled manner. While graph-based methodologies have proven effective for multi-view learning, the systematic development of optimization-centric methodologies for consistent graph fusion remains in its nascent stages. Furthermore, the lack of explicit alignment between the latent feature representation space and the graph space is frequently overlooked, ultimately compromising the consistency of the shared representation. To this end, we propose a bi-level framework for joint multi-view feature representation learning and graph fusion, which formally formulates the multi-view data fusion task as a rigorous mathematical optimization problem. The proposed approach demonstrates consistent effectiveness in clustering tasks, achieving state-of-the-art performance on six benchmark datasets with accuracy improvements ranging up to 9.73% over existing methods.</description>
	<pubDate>2026-08-27</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 289: Learning Multi-View Representations with Graph-Level Fusion: A Bi-Level Optimization Perspective</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/9/289">doi: 10.3390/bdcc10090289</a></p>
	<p>Authors:
		Na Song
		Xin Wang
		Zhicai Zhang
		Shiping Wang
		</p>
	<p>Multi-view representation learning, aiming to extract discriminative patterns from heterogeneous data sources, has attracted increasing attention in recent years. A central challenge in multi-view data fusion lies in determining the appropriate weight and importance of each view in a principled manner. While graph-based methodologies have proven effective for multi-view learning, the systematic development of optimization-centric methodologies for consistent graph fusion remains in its nascent stages. Furthermore, the lack of explicit alignment between the latent feature representation space and the graph space is frequently overlooked, ultimately compromising the consistency of the shared representation. To this end, we propose a bi-level framework for joint multi-view feature representation learning and graph fusion, which formally formulates the multi-view data fusion task as a rigorous mathematical optimization problem. The proposed approach demonstrates consistent effectiveness in clustering tasks, achieving state-of-the-art performance on six benchmark datasets with accuracy improvements ranging up to 9.73% over existing methods.</p>
	]]></content:encoded>

	<dc:title>Learning Multi-View Representations with Graph-Level Fusion: A Bi-Level Optimization Perspective</dc:title>
			<dc:creator>Na Song</dc:creator>
			<dc:creator>Xin Wang</dc:creator>
			<dc:creator>Zhicai Zhang</dc:creator>
			<dc:creator>Shiping Wang</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10090289</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-08-27</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-08-27</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>9</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>289</prism:startingPage>
		<prism:doi>10.3390/bdcc10090289</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/9/289</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/9/288">

	<title>BDCC, Vol. 10, Pages 288: On the Information Retention Capabilities of Hierarchical JSON Representations of Scientific Texts</title>
	<link>https://www.mdpi.com/2504-2289/10/9/288</link>
	<description>This paper studies whether structured representations can retain the meaning contained in scientific sentences. A lightweight LLM is fine-tuned to generate JSON structures from scientific sentences. The objective was to capture information contained in the sentence using a hierarchical structure. The structured representations are then used to reconstruct sentences with a generative model to test whether such representations are useful to retain the meaning of the sentence. The evaluation compares the reconstructed sentences with the originals using semantic and lexical similarity. Our results show that hierarchical structured formats preserve semantic information with an average cosine similarity of 0.87.</description>
	<pubDate>2026-08-27</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 288: On the Information Retention Capabilities of Hierarchical JSON Representations of Scientific Texts</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/9/288">doi: 10.3390/bdcc10090288</a></p>
	<p>Authors:
		Satya Sri Rajiteswari Nimmagadda
		Ethan Young
		Niladri Sengupta
		Ananya Jana
		Aniruddha Maiti
		</p>
	<p>This paper studies whether structured representations can retain the meaning contained in scientific sentences. A lightweight LLM is fine-tuned to generate JSON structures from scientific sentences. The objective was to capture information contained in the sentence using a hierarchical structure. The structured representations are then used to reconstruct sentences with a generative model to test whether such representations are useful to retain the meaning of the sentence. The evaluation compares the reconstructed sentences with the originals using semantic and lexical similarity. Our results show that hierarchical structured formats preserve semantic information with an average cosine similarity of 0.87.</p>
	]]></content:encoded>

	<dc:title>On the Information Retention Capabilities of Hierarchical JSON Representations of Scientific Texts</dc:title>
			<dc:creator>Satya Sri Rajiteswari Nimmagadda</dc:creator>
			<dc:creator>Ethan Young</dc:creator>
			<dc:creator>Niladri Sengupta</dc:creator>
			<dc:creator>Ananya Jana</dc:creator>
			<dc:creator>Aniruddha Maiti</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10090288</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-08-27</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-08-27</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>9</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>288</prism:startingPage>
		<prism:doi>10.3390/bdcc10090288</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/9/288</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/9/287">

	<title>BDCC, Vol. 10, Pages 287: Edge-Geometry-Guided Deformable Detection for Sub-Millimeter Defects in Underwater Nuclear Component Inspection</title>
	<link>https://www.mdpi.com/2504-2289/10/9/287</link>
	<description>Accurate detection of sub-millimeter defects in reactor core-plate cotter-pin holes is essential for nuclear safety. However, underwater inspection images often suffer from low signal-to-noise ratios, weak boundary responses, and pseudo-edge interference, resulting in unstable localization of defects. Existing deformable and attention-based detectors remain vulnerable to sampling drift and semantic&amp;amp;ndash;boundary inconsistency under such conditions. To address these challenges, an Edge-Geometry-Guided Deformable Detection Network (EGD-Net) is proposed for underwater defect detection. EGD-Net introduces an edge-geometry-constrained deformable sampling mechanism that embeds edge-confidence priors into deformable convolution to improve boundary-aware feature sampling. A cross-level semantic&amp;amp;ndash;geometric alignment strategy is designed to enhance the interaction between defect semantics and geometric boundary cues, while a top-down feedback recalibration mechanism improves multi-scale response consistency for weak defects. Experiments on the Core-Plate Pin-Hole Defect (CPHD) dataset demonstrate that EGD-Net achieves the highest AP@[0.5:0.95] on both datasets while maintaining competitive or superior Precision, Recall, and F1-score while reducing engineering center error under a fixed operating point. Performance across the two complementary domains suggests its robustness to variations between coupon images and practical underwater inspection scenes. These results indicate that EGD-Net provides a reliable solution for boundary-sensitive localization of underwater sub-millimeter defects in nuclear inspection.</description>
	<pubDate>2026-08-26</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 287: Edge-Geometry-Guided Deformable Detection for Sub-Millimeter Defects in Underwater Nuclear Component Inspection</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/9/287">doi: 10.3390/bdcc10090287</a></p>
	<p>Authors:
		Jinkun Li
		Lingyu Sun
		Minglu Zhang
		Chao Ma
		Xinbao Li
		</p>
	<p>Accurate detection of sub-millimeter defects in reactor core-plate cotter-pin holes is essential for nuclear safety. However, underwater inspection images often suffer from low signal-to-noise ratios, weak boundary responses, and pseudo-edge interference, resulting in unstable localization of defects. Existing deformable and attention-based detectors remain vulnerable to sampling drift and semantic&amp;amp;ndash;boundary inconsistency under such conditions. To address these challenges, an Edge-Geometry-Guided Deformable Detection Network (EGD-Net) is proposed for underwater defect detection. EGD-Net introduces an edge-geometry-constrained deformable sampling mechanism that embeds edge-confidence priors into deformable convolution to improve boundary-aware feature sampling. A cross-level semantic&amp;amp;ndash;geometric alignment strategy is designed to enhance the interaction between defect semantics and geometric boundary cues, while a top-down feedback recalibration mechanism improves multi-scale response consistency for weak defects. Experiments on the Core-Plate Pin-Hole Defect (CPHD) dataset demonstrate that EGD-Net achieves the highest AP@[0.5:0.95] on both datasets while maintaining competitive or superior Precision, Recall, and F1-score while reducing engineering center error under a fixed operating point. Performance across the two complementary domains suggests its robustness to variations between coupon images and practical underwater inspection scenes. These results indicate that EGD-Net provides a reliable solution for boundary-sensitive localization of underwater sub-millimeter defects in nuclear inspection.</p>
	]]></content:encoded>

	<dc:title>Edge-Geometry-Guided Deformable Detection for Sub-Millimeter Defects in Underwater Nuclear Component Inspection</dc:title>
			<dc:creator>Jinkun Li</dc:creator>
			<dc:creator>Lingyu Sun</dc:creator>
			<dc:creator>Minglu Zhang</dc:creator>
			<dc:creator>Chao Ma</dc:creator>
			<dc:creator>Xinbao Li</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10090287</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-08-26</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-08-26</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>9</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>287</prism:startingPage>
		<prism:doi>10.3390/bdcc10090287</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/9/287</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/9/286">

	<title>BDCC, Vol. 10, Pages 286: Cloud&amp;ndash;Edge Intelligence for the Industrial Internet of Things: Security, Digital Twins, and Resource-Aware Computing</title>
	<link>https://www.mdpi.com/2504-2289/10/9/286</link>
	<description>The Industrial Internet of Things (IIoT) is reshaping the relationship between physical industrial processes and digital computing infrastructures [...]</description>
	<pubDate>2026-08-26</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 286: Cloud&amp;ndash;Edge Intelligence for the Industrial Internet of Things: Security, Digital Twins, and Resource-Aware Computing</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/9/286">doi: 10.3390/bdcc10090286</a></p>
	<p>Authors:
		Muhammad Kazim
		Stefan Kuhn
		Mujeeb Ur Rehman
		</p>
	<p>The Industrial Internet of Things (IIoT) is reshaping the relationship between physical industrial processes and digital computing infrastructures [...]</p>
	]]></content:encoded>

	<dc:title>Cloud&amp;amp;ndash;Edge Intelligence for the Industrial Internet of Things: Security, Digital Twins, and Resource-Aware Computing</dc:title>
			<dc:creator>Muhammad Kazim</dc:creator>
			<dc:creator>Stefan Kuhn</dc:creator>
			<dc:creator>Mujeeb Ur Rehman</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10090286</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-08-26</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-08-26</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>9</prism:number>
	<prism:section>Editorial</prism:section>
	<prism:startingPage>286</prism:startingPage>
		<prism:doi>10.3390/bdcc10090286</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/9/286</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/9/285">

	<title>BDCC, Vol. 10, Pages 285: EvoPhySec: A Multi-Objective Evolutionary Framework for Adaptive Calibration of Physics-Informed Anomaly Detection in Smart City IoT Networks</title>
	<link>https://www.mdpi.com/2504-2289/10/9/285</link>
	<description>Physics-informed anomaly detectors on smart city IoT gateways face a conflict that static calibration cannot solve. Higher detection accuracy (F1) costs inference latency and edge energy. The trade-off differs across urban sensor domains. PhySec-Edge showed that hybrid physics-informed edge-AI detection works in industrial IoT. Its calibration is static. Transferred to smart city verticals with different physical constraints, it degrades. EvoPhySec addresses this problem by introducing a multi-objective evolutionary calibration layer that adapts PhySec-Edge to five heterogeneous smart city verticals under simultaneous accuracy, latency, and energy constraints. We treat the Physics Validation Engine (PVE) thresholds, the Edge AI Detection Engine (EADE) ensemble weights, and the decision rule parameters as one three-objective optimization problem. It is solved with NSGA-III plus the Physics-Constraint Preservation Operator (PCPO), which keeps candidate solutions physically feasible during the search. Evaluated on a synthetic multi-domain smart city dataset (n = 47,250 samples, five urban verticals, eleven attack classes, five random seeds) and validated on BATADAL and a CIC-IoT2023-inspired benchmark, EvoPhySec achieves mean F1 = 0.847 &amp;amp;plusmn; 0.004 at 34.2 ms latency and 0.71 W edge power, Pareto-dominating the static baseline on all three objectives. The ablation study isolates the PCPO contribution at +0.031 HVI over unconstrained NSGA-III. Cross-domain transfer analysis identifies a three-vertical transferable cluster and confirms that critical infra-structure requires domain-specific calibration. These results are demonstrated on synthetic data; real-world multi-vertical validation remains future work.</description>
	<pubDate>2026-08-25</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 285: EvoPhySec: A Multi-Objective Evolutionary Framework for Adaptive Calibration of Physics-Informed Anomaly Detection in Smart City IoT Networks</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/9/285">doi: 10.3390/bdcc10090285</a></p>
	<p>Authors:
		Dalibor Radovanovic
		Nikola Savanovic
		Jelena Janackovic
		Petar Kresoja
		Bojan Papaz
		</p>
	<p>Physics-informed anomaly detectors on smart city IoT gateways face a conflict that static calibration cannot solve. Higher detection accuracy (F1) costs inference latency and edge energy. The trade-off differs across urban sensor domains. PhySec-Edge showed that hybrid physics-informed edge-AI detection works in industrial IoT. Its calibration is static. Transferred to smart city verticals with different physical constraints, it degrades. EvoPhySec addresses this problem by introducing a multi-objective evolutionary calibration layer that adapts PhySec-Edge to five heterogeneous smart city verticals under simultaneous accuracy, latency, and energy constraints. We treat the Physics Validation Engine (PVE) thresholds, the Edge AI Detection Engine (EADE) ensemble weights, and the decision rule parameters as one three-objective optimization problem. It is solved with NSGA-III plus the Physics-Constraint Preservation Operator (PCPO), which keeps candidate solutions physically feasible during the search. Evaluated on a synthetic multi-domain smart city dataset (n = 47,250 samples, five urban verticals, eleven attack classes, five random seeds) and validated on BATADAL and a CIC-IoT2023-inspired benchmark, EvoPhySec achieves mean F1 = 0.847 &amp;amp;plusmn; 0.004 at 34.2 ms latency and 0.71 W edge power, Pareto-dominating the static baseline on all three objectives. The ablation study isolates the PCPO contribution at +0.031 HVI over unconstrained NSGA-III. Cross-domain transfer analysis identifies a three-vertical transferable cluster and confirms that critical infra-structure requires domain-specific calibration. These results are demonstrated on synthetic data; real-world multi-vertical validation remains future work.</p>
	]]></content:encoded>

	<dc:title>EvoPhySec: A Multi-Objective Evolutionary Framework for Adaptive Calibration of Physics-Informed Anomaly Detection in Smart City IoT Networks</dc:title>
			<dc:creator>Dalibor Radovanovic</dc:creator>
			<dc:creator>Nikola Savanovic</dc:creator>
			<dc:creator>Jelena Janackovic</dc:creator>
			<dc:creator>Petar Kresoja</dc:creator>
			<dc:creator>Bojan Papaz</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10090285</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-08-25</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-08-25</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>9</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>285</prism:startingPage>
		<prism:doi>10.3390/bdcc10090285</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/9/285</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/9/284">

	<title>BDCC, Vol. 10, Pages 284: Socioenvironmental Vulnerability Profiles and Health Expenditure in Mexican Households Using Data Science</title>
	<link>https://www.mdpi.com/2504-2289/10/9/284</link>
	<description>This study aimed to identify socioenvironmental vulnerability profiles among Mexican households and analyze their association with health expenditure. Data from 86,102 households included in the 2024 National Household Income and Expenditure Survey were analyzed. Socioenvironmental profiles were constructed using housing, basic services, sanitation, household energy, socioeconomic stratum, and overcrowding through factor analysis of mixed data and k-means. Internal validation, stability analyses, algorithm comparisons, and sensitivity analyses supported a three-profile solution representing low, intermediate, and high vulnerability. Health expenditure was examined using survey-weighted descriptive estimates, exploratory nonparametric comparisons, and a survey-adjusted two-part model. The intermediate vulnerability profile showed higher odds of reporting health expenditure than the low vulnerability profile (OR = 1.129, 95% CI: 1.048 to 1.217, p = 0.002), whereas the high vulnerability profile showed no significant difference. Among households with positive expenditure, high vulnerability was associated with lower logarithmic expenditure (&amp;amp;beta;=&amp;amp;minus;0.341, 95% CI: &amp;amp;minus;0.490 to &amp;amp;minus;0.192, p &amp;amp;lt; 0.001), whereas the intermediate profile was not significantly different in the main model. The association for high vulnerability remained significant in sensitivity analyses, while results for the intermediate profile were more sensitive to model specification and income adjustment. Despite several statistically significant associations, effect sizes in the exploratory comparisons were small and the regression models explained a limited proportion of the variability in health expenditure. The findings therefore indicate modest associations between socioenvironmental vulnerability and health expenditure, with household economic resources and other unmeasured health-related factors likely contributing to the observed differences.</description>
	<pubDate>2026-08-25</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 284: Socioenvironmental Vulnerability Profiles and Health Expenditure in Mexican Households Using Data Science</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/9/284">doi: 10.3390/bdcc10090284</a></p>
	<p>Authors:
		Héctor Alejandro Acuña-Cid
		Eduardo Ahumada-Tello
		Cristina Almeida-Perales
		Mónica Judith Chávez-Soto
		Pablo Gerardo Guerrero-Herrera
		José Eduardo Briceño-Muro
		</p>
	<p>This study aimed to identify socioenvironmental vulnerability profiles among Mexican households and analyze their association with health expenditure. Data from 86,102 households included in the 2024 National Household Income and Expenditure Survey were analyzed. Socioenvironmental profiles were constructed using housing, basic services, sanitation, household energy, socioeconomic stratum, and overcrowding through factor analysis of mixed data and k-means. Internal validation, stability analyses, algorithm comparisons, and sensitivity analyses supported a three-profile solution representing low, intermediate, and high vulnerability. Health expenditure was examined using survey-weighted descriptive estimates, exploratory nonparametric comparisons, and a survey-adjusted two-part model. The intermediate vulnerability profile showed higher odds of reporting health expenditure than the low vulnerability profile (OR = 1.129, 95% CI: 1.048 to 1.217, p = 0.002), whereas the high vulnerability profile showed no significant difference. Among households with positive expenditure, high vulnerability was associated with lower logarithmic expenditure (&amp;amp;beta;=&amp;amp;minus;0.341, 95% CI: &amp;amp;minus;0.490 to &amp;amp;minus;0.192, p &amp;amp;lt; 0.001), whereas the intermediate profile was not significantly different in the main model. The association for high vulnerability remained significant in sensitivity analyses, while results for the intermediate profile were more sensitive to model specification and income adjustment. Despite several statistically significant associations, effect sizes in the exploratory comparisons were small and the regression models explained a limited proportion of the variability in health expenditure. The findings therefore indicate modest associations between socioenvironmental vulnerability and health expenditure, with household economic resources and other unmeasured health-related factors likely contributing to the observed differences.</p>
	]]></content:encoded>

	<dc:title>Socioenvironmental Vulnerability Profiles and Health Expenditure in Mexican Households Using Data Science</dc:title>
			<dc:creator>Héctor Alejandro Acuña-Cid</dc:creator>
			<dc:creator>Eduardo Ahumada-Tello</dc:creator>
			<dc:creator>Cristina Almeida-Perales</dc:creator>
			<dc:creator>Mónica Judith Chávez-Soto</dc:creator>
			<dc:creator>Pablo Gerardo Guerrero-Herrera</dc:creator>
			<dc:creator>José Eduardo Briceño-Muro</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10090284</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-08-25</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-08-25</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>9</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>284</prism:startingPage>
		<prism:doi>10.3390/bdcc10090284</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/9/284</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/9/283">

	<title>BDCC, Vol. 10, Pages 283: Time-Gated Multi-Expert Generative Adversarial Network for Gearbox Fault Diagnosis</title>
	<link>https://www.mdpi.com/2504-2289/10/9/283</link>
	<description>In the domain of rotating machinery fault diagnosis, challenges such as multi-operating condition distribution heterogeneity and the difficulty of distinguishing fault features within multi-scale temporal signals persist. To address these issues, this paper introduces the Time-Gated Multi-Expert Generative Adversarial Network (TGME-GAN), a fault diagnosis approach that integrates a multi-expert gated conditional generative adversarial network with a clustering structure-aware feature enhancement. This method combines unsupervised K-means clustering with supervised discriminative learning. The optimal number of clusters is selected adaptively using the silhouette coefficient, and the distance vector from each sample to the cluster centers serves as a topological prior feature. A spatial&amp;amp;ndash;temporal joint representation matrix is then formed by concatenating PCA principal components, differential features, cumulative statistical features, and standardized change rates, which together capture both abrupt mutations and progressive degradation in fault signals. In the model, the discriminator incorporates a multi-expert gated network. Each expert learns a feature subspace corresponding to a distinct operating condition, and the gated network dynamically assigns fusion weights, allowing the discriminator to capture heterogeneous distributions across industrial conditions. The generator extracts multi-scale local patterns with a three-layer one-dimensional convolutional network and models sequential dependencies with a two-layer LSTM, producing high-quality fault samples that preserve intrinsic consistency. At the engineering level, TGME-GAN is deployed for gearbox fault diagnosis in uneven, small-sample industrial settings. In two gearbox fault experiments, this method substantially outperforms current mainstream models.</description>
	<pubDate>2026-08-22</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 283: Time-Gated Multi-Expert Generative Adversarial Network for Gearbox Fault Diagnosis</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/9/283">doi: 10.3390/bdcc10090283</a></p>
	<p>Authors:
		Puyang Guan
		Zhe Wei
		Lei Wang
		Lang Lang
		</p>
	<p>In the domain of rotating machinery fault diagnosis, challenges such as multi-operating condition distribution heterogeneity and the difficulty of distinguishing fault features within multi-scale temporal signals persist. To address these issues, this paper introduces the Time-Gated Multi-Expert Generative Adversarial Network (TGME-GAN), a fault diagnosis approach that integrates a multi-expert gated conditional generative adversarial network with a clustering structure-aware feature enhancement. This method combines unsupervised K-means clustering with supervised discriminative learning. The optimal number of clusters is selected adaptively using the silhouette coefficient, and the distance vector from each sample to the cluster centers serves as a topological prior feature. A spatial&amp;amp;ndash;temporal joint representation matrix is then formed by concatenating PCA principal components, differential features, cumulative statistical features, and standardized change rates, which together capture both abrupt mutations and progressive degradation in fault signals. In the model, the discriminator incorporates a multi-expert gated network. Each expert learns a feature subspace corresponding to a distinct operating condition, and the gated network dynamically assigns fusion weights, allowing the discriminator to capture heterogeneous distributions across industrial conditions. The generator extracts multi-scale local patterns with a three-layer one-dimensional convolutional network and models sequential dependencies with a two-layer LSTM, producing high-quality fault samples that preserve intrinsic consistency. At the engineering level, TGME-GAN is deployed for gearbox fault diagnosis in uneven, small-sample industrial settings. In two gearbox fault experiments, this method substantially outperforms current mainstream models.</p>
	]]></content:encoded>

	<dc:title>Time-Gated Multi-Expert Generative Adversarial Network for Gearbox Fault Diagnosis</dc:title>
			<dc:creator>Puyang Guan</dc:creator>
			<dc:creator>Zhe Wei</dc:creator>
			<dc:creator>Lei Wang</dc:creator>
			<dc:creator>Lang Lang</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10090283</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-08-22</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-08-22</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>9</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>283</prism:startingPage>
		<prism:doi>10.3390/bdcc10090283</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/9/283</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/8/282">

	<title>BDCC, Vol. 10, Pages 282: Atmospheric Turbulence Mitigation in the Deep Learning Era: A Critical Review from CNNs and GANs to Transformers, Diffusion, Mamba, and Physics-Informed Models</title>
	<link>https://www.mdpi.com/2504-2289/10/8/282</link>
	<description>Anyone who has watched a distant scene shimmer above hot pavement has seen atmospheric turbulence destroy image detail. In long-range imaging, turbulence produces spatially varying blur, geometric warping, scintillation, and temporal instability. Recovering the underlying scene is therefore an ill-posed inverse problem, and learned priors must compensate for distortions that simplified optical models capture only partially. We use turbulence mitigation as the umbrella term for all countermeasures and turbulence restoration for its computational core, the estimation of a clean image from degraded observations. From a cognitive-computing perspective, mitigation is not merely image enhancement. It is an uncertainty-constrained visual inference problem in which an intelligent system must reconstruct, interpret, and act on observations relayed through a stochastic physical channel. This critical review examines how the deep learning era has reshaped turbulence mitigation, with physics as the foundation for understanding degradation and designing inductive biases. It traces the architectural progression from convolutional and adversarial networks to Transformers, denoising diffusion models, Mamba and other state-space architectures, and physics-informed frameworks. For each family, we ask a common question: How does it treat the aleatoric uncertainty intrinsic to a random optical channel and the epistemic uncertainty introduced by scarce and simulator-dominated training data? The review also analyzes datasets, simulation strategies, loss functions, and evaluation metrics, and it separates the small body of shared-protocol benchmark evidence from the far larger body of self-reported results that cannot be compared across studies. Persistent obstacles include the synthetic-to-real domain gap, the scarcity of paired real turbulence data, the mismatch between fidelity metrics and downstream task performance, and the computational cost that limits operational deployment. We close with an AI-centered agenda in which uncertainty quantification stands alongside domain adaptation as a first-order priority.</description>
	<pubDate>2026-08-21</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 282: Atmospheric Turbulence Mitigation in the Deep Learning Era: A Critical Review from CNNs and GANs to Transformers, Diffusion, Mamba, and Physics-Informed Models</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/8/282">doi: 10.3390/bdcc10080282</a></p>
	<p>Authors:
		Nurul Jannah
		Teddy Surya Gunawan
		Mira Kartiwi
		Nadirah Abdul Rahim
		Ali Sophian
		</p>
	<p>Anyone who has watched a distant scene shimmer above hot pavement has seen atmospheric turbulence destroy image detail. In long-range imaging, turbulence produces spatially varying blur, geometric warping, scintillation, and temporal instability. Recovering the underlying scene is therefore an ill-posed inverse problem, and learned priors must compensate for distortions that simplified optical models capture only partially. We use turbulence mitigation as the umbrella term for all countermeasures and turbulence restoration for its computational core, the estimation of a clean image from degraded observations. From a cognitive-computing perspective, mitigation is not merely image enhancement. It is an uncertainty-constrained visual inference problem in which an intelligent system must reconstruct, interpret, and act on observations relayed through a stochastic physical channel. This critical review examines how the deep learning era has reshaped turbulence mitigation, with physics as the foundation for understanding degradation and designing inductive biases. It traces the architectural progression from convolutional and adversarial networks to Transformers, denoising diffusion models, Mamba and other state-space architectures, and physics-informed frameworks. For each family, we ask a common question: How does it treat the aleatoric uncertainty intrinsic to a random optical channel and the epistemic uncertainty introduced by scarce and simulator-dominated training data? The review also analyzes datasets, simulation strategies, loss functions, and evaluation metrics, and it separates the small body of shared-protocol benchmark evidence from the far larger body of self-reported results that cannot be compared across studies. Persistent obstacles include the synthetic-to-real domain gap, the scarcity of paired real turbulence data, the mismatch between fidelity metrics and downstream task performance, and the computational cost that limits operational deployment. We close with an AI-centered agenda in which uncertainty quantification stands alongside domain adaptation as a first-order priority.</p>
	]]></content:encoded>

	<dc:title>Atmospheric Turbulence Mitigation in the Deep Learning Era: A Critical Review from CNNs and GANs to Transformers, Diffusion, Mamba, and Physics-Informed Models</dc:title>
			<dc:creator>Nurul Jannah</dc:creator>
			<dc:creator>Teddy Surya Gunawan</dc:creator>
			<dc:creator>Mira Kartiwi</dc:creator>
			<dc:creator>Nadirah Abdul Rahim</dc:creator>
			<dc:creator>Ali Sophian</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10080282</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-08-21</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-08-21</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>8</prism:number>
	<prism:section>Review</prism:section>
	<prism:startingPage>282</prism:startingPage>
		<prism:doi>10.3390/bdcc10080282</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/8/282</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/8/281">

	<title>BDCC, Vol. 10, Pages 281: Predicting Recurring Treatment Events Within Multiple Future Time Windows</title>
	<link>https://www.mdpi.com/2504-2289/10/8/281</link>
	<description>Medical treatment decision making is a complex process that involves integrating multivariate time-oriented data from multiple sources and is often influenced by factors such as patient load. In this study, we propose the Recurring Target Prediction (RTP) Pipeline to support treatment decision making by predicting the next medical action most likely to be administered, based on the historical data from patients in similar contexts. The method transforms raw time-stamped data into symbolic time intervals, incorporating domain knowledge. Each of the patient&amp;amp;rsquo;s data are segmented by pre-defined trigger conditions (e.g., hypoglycemia), with each segment containing a feature window (historical data as symbolic time intervals); a prediction window (e.g., treatment dosage); and an optional prediction gap between the feature and prediction windows, enabling a future treatment alert. A frequent pattern-mining method is applied to the feature windows, and features generated from the mined patterns (e.g., count within each record and mean duration) are used as input to a Two-Step prediction model. First, a binary classifier predicts whether treatment is necessary, followed by a regression model to predict dosage. Finally, SHapley Additive exPlanations (SHAP) provide insights into the model&amp;amp;rsquo;s decision making. We have evaluated the pipeline on an Intensive Care Unit (ICU) dataset, across three domains: hypoglycemia, hypokalemia, and hypotension. Key contributions include leveraging the recurrence of medical conditions and events to enrich the dataset, reducing false positives through a Two-Step prediction model, allowing prediction gaps for advance treatment notice, and incorporating SHAP, and introducing a two-level SHAP-based method for aggregating the relative weights of temporal patterns and components, to enhance the model&amp;amp;rsquo;s interpretability.</description>
	<pubDate>2026-08-21</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 281: Predicting Recurring Treatment Events Within Multiple Future Time Windows</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/8/281">doi: 10.3390/bdcc10080281</a></p>
	<p>Authors:
		Michal Weisman Raymond
		Yuval Shahar
		</p>
	<p>Medical treatment decision making is a complex process that involves integrating multivariate time-oriented data from multiple sources and is often influenced by factors such as patient load. In this study, we propose the Recurring Target Prediction (RTP) Pipeline to support treatment decision making by predicting the next medical action most likely to be administered, based on the historical data from patients in similar contexts. The method transforms raw time-stamped data into symbolic time intervals, incorporating domain knowledge. Each of the patient&amp;amp;rsquo;s data are segmented by pre-defined trigger conditions (e.g., hypoglycemia), with each segment containing a feature window (historical data as symbolic time intervals); a prediction window (e.g., treatment dosage); and an optional prediction gap between the feature and prediction windows, enabling a future treatment alert. A frequent pattern-mining method is applied to the feature windows, and features generated from the mined patterns (e.g., count within each record and mean duration) are used as input to a Two-Step prediction model. First, a binary classifier predicts whether treatment is necessary, followed by a regression model to predict dosage. Finally, SHapley Additive exPlanations (SHAP) provide insights into the model&amp;amp;rsquo;s decision making. We have evaluated the pipeline on an Intensive Care Unit (ICU) dataset, across three domains: hypoglycemia, hypokalemia, and hypotension. Key contributions include leveraging the recurrence of medical conditions and events to enrich the dataset, reducing false positives through a Two-Step prediction model, allowing prediction gaps for advance treatment notice, and incorporating SHAP, and introducing a two-level SHAP-based method for aggregating the relative weights of temporal patterns and components, to enhance the model&amp;amp;rsquo;s interpretability.</p>
	]]></content:encoded>

	<dc:title>Predicting Recurring Treatment Events Within Multiple Future Time Windows</dc:title>
			<dc:creator>Michal Weisman Raymond</dc:creator>
			<dc:creator>Yuval Shahar</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10080281</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-08-21</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-08-21</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>8</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>281</prism:startingPage>
		<prism:doi>10.3390/bdcc10080281</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/8/281</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/8/280">

	<title>BDCC, Vol. 10, Pages 280: Subject-Specific BCI Frameworks for Motor Imagery Classification Based on BSS-Free and BSS-Equipped Pipelines</title>
	<link>https://www.mdpi.com/2504-2289/10/8/280</link>
	<description>The development of Motor Imagery (MI) Brain&amp;amp;ndash;Computer Interfaces (BCIs) is systematically constrained by low signal-to-noise ratios (SNRs), signal non-stationarity, and acute data scarcity. While complex Blind Source Separation (BSS) methods optimize signal clarity, their computational overhead introduces propagation delays that challenge real-time constraints. This study addresses this engineering trade-off by introducing a localized architectural framework to evaluate whether a lightweight pipeline operating without BSS (No-BSS) is sufficiently efficient for real-time control when compared against two BSS-equipped pipelines utilizing Independent Component Analysis (ICA) and Empirical Mode Decomposition (EMD). Validated across the BCI Competition IV Dataset 2A and the PhysioNet MI dataset, all three pipelines share an identical processing chain designed to maximize efficiency. To mitigate low SNRs, an Adaptive Laplacian spatial filter isolates neural intent across target sensorimotor electrodes (C3, C4, and Cz). Data scarcity is countered via a Gaussian noise injection data augmentation strategy, while session-to-session variability is addressed during feature extraction using Wavelet Packet Decomposition (WPD) paired with a Fisher Score criterion to dynamically isolate subject-specific time-frequency nodes. Redundant features are subsequently eliminated using a Genetic Algorithm (GA) before classification. Experimental evaluation reveals a distinct performance stratification: while the ICA (92.80%) and EMD (92.69%) pipelines yield the highest average accuracy for the PhysioNet dataset by isolating non-stationary and physiological noise, the No-BSS baseline (90.28%) remains the superior framework for the BCI Dataset 2A. Across all pipelines across both datasets, a stable classification hierarchy emerges wherein the Support Vector Machine (SVM) leads performance due to its maximum-margin decision boundary, followed by k-Nearest Neighbors (kNN), a modified EEGNet, and Decision Trees. The No-BSS baseline achieves classification accuracies highly competitive with its BSS counterparts while entirely bypassing their algorithmic overhead. Given the strict latency constraints of live BCI control loops, these findings establish the optimized No-BSS pipeline as a highly viable alternative for low-latency, real-time implementations.</description>
	<pubDate>2026-08-20</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 280: Subject-Specific BCI Frameworks for Motor Imagery Classification Based on BSS-Free and BSS-Equipped Pipelines</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/8/280">doi: 10.3390/bdcc10080280</a></p>
	<p>Authors:
		Nerita Ramsoonder
		Rito Clifford Maswanganyi
		Philani Khumalo
		</p>
	<p>The development of Motor Imagery (MI) Brain&amp;amp;ndash;Computer Interfaces (BCIs) is systematically constrained by low signal-to-noise ratios (SNRs), signal non-stationarity, and acute data scarcity. While complex Blind Source Separation (BSS) methods optimize signal clarity, their computational overhead introduces propagation delays that challenge real-time constraints. This study addresses this engineering trade-off by introducing a localized architectural framework to evaluate whether a lightweight pipeline operating without BSS (No-BSS) is sufficiently efficient for real-time control when compared against two BSS-equipped pipelines utilizing Independent Component Analysis (ICA) and Empirical Mode Decomposition (EMD). Validated across the BCI Competition IV Dataset 2A and the PhysioNet MI dataset, all three pipelines share an identical processing chain designed to maximize efficiency. To mitigate low SNRs, an Adaptive Laplacian spatial filter isolates neural intent across target sensorimotor electrodes (C3, C4, and Cz). Data scarcity is countered via a Gaussian noise injection data augmentation strategy, while session-to-session variability is addressed during feature extraction using Wavelet Packet Decomposition (WPD) paired with a Fisher Score criterion to dynamically isolate subject-specific time-frequency nodes. Redundant features are subsequently eliminated using a Genetic Algorithm (GA) before classification. Experimental evaluation reveals a distinct performance stratification: while the ICA (92.80%) and EMD (92.69%) pipelines yield the highest average accuracy for the PhysioNet dataset by isolating non-stationary and physiological noise, the No-BSS baseline (90.28%) remains the superior framework for the BCI Dataset 2A. Across all pipelines across both datasets, a stable classification hierarchy emerges wherein the Support Vector Machine (SVM) leads performance due to its maximum-margin decision boundary, followed by k-Nearest Neighbors (kNN), a modified EEGNet, and Decision Trees. The No-BSS baseline achieves classification accuracies highly competitive with its BSS counterparts while entirely bypassing their algorithmic overhead. Given the strict latency constraints of live BCI control loops, these findings establish the optimized No-BSS pipeline as a highly viable alternative for low-latency, real-time implementations.</p>
	]]></content:encoded>

	<dc:title>Subject-Specific BCI Frameworks for Motor Imagery Classification Based on BSS-Free and BSS-Equipped Pipelines</dc:title>
			<dc:creator>Nerita Ramsoonder</dc:creator>
			<dc:creator>Rito Clifford Maswanganyi</dc:creator>
			<dc:creator>Philani Khumalo</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10080280</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-08-20</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-08-20</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>8</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>280</prism:startingPage>
		<prism:doi>10.3390/bdcc10080280</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/8/280</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/8/279">

	<title>BDCC, Vol. 10, Pages 279: TE-FEDformer: A Time-Series-Enhanced FEDformer for Remaining Useful Life Prediction of Rolling Bearings</title>
	<link>https://www.mdpi.com/2504-2289/10/8/279</link>
	<description>In the era of intelligence, accurate remaining useful life (RUL) prediction is essential to ensure the reliable operation of smart equipment, particularly for rolling bearings&amp;amp;mdash;critical components that are highly susceptible to degradation in rotating machinery. However, as faults progressively develop, the vibration signals of rolling bearings exhibit strong non-stationarity and complex degradation patterns. Existing RUL prediction methods, particularly standard Transformer-based models, often struggle to capture local transient features within non-stationary signals and fail to effectively decouple long-term degradation trends from periodic variations. To overcome these limitations, a novel RUL prediction method that integrates time-series analysis techniques with the FEDformer architecture is proposed, termed TE-FEDformer. Firstly, a feature enhancement module is employed at the input stage to reconstruct and strengthen the original sequence, aiming to strengthen the representation of weak fault features that are often overlooked by global attention mechanisms. Then, deep time-series representations are extracted via the encoder. In the decoding stage, a frequency enhancement mechanism and a sequence decomposition mechanism are jointly utilized to explicitly model the coupling between degradation trends and periodic variations, thus resolving the spectral interference commonly encountered in complex degradation processes. Comparative experimental results on the PHM2012 and XJTU-SY datasets demonstrate that TE-FEDformer outperforms other benchmark models. Ablation studies further validate that each module contributes positively to the overall performance, confirming the effectiveness of the proposed approach for RUL prediction.</description>
	<pubDate>2026-08-18</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 279: TE-FEDformer: A Time-Series-Enhanced FEDformer for Remaining Useful Life Prediction of Rolling Bearings</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/8/279">doi: 10.3390/bdcc10080279</a></p>
	<p>Authors:
		Yazhou Zhou
		Mingyang Tang
		Yunzhu Shan
		Wenbo Wang
		Man Zhou
		Yuchun Peng
		</p>
	<p>In the era of intelligence, accurate remaining useful life (RUL) prediction is essential to ensure the reliable operation of smart equipment, particularly for rolling bearings&amp;amp;mdash;critical components that are highly susceptible to degradation in rotating machinery. However, as faults progressively develop, the vibration signals of rolling bearings exhibit strong non-stationarity and complex degradation patterns. Existing RUL prediction methods, particularly standard Transformer-based models, often struggle to capture local transient features within non-stationary signals and fail to effectively decouple long-term degradation trends from periodic variations. To overcome these limitations, a novel RUL prediction method that integrates time-series analysis techniques with the FEDformer architecture is proposed, termed TE-FEDformer. Firstly, a feature enhancement module is employed at the input stage to reconstruct and strengthen the original sequence, aiming to strengthen the representation of weak fault features that are often overlooked by global attention mechanisms. Then, deep time-series representations are extracted via the encoder. In the decoding stage, a frequency enhancement mechanism and a sequence decomposition mechanism are jointly utilized to explicitly model the coupling between degradation trends and periodic variations, thus resolving the spectral interference commonly encountered in complex degradation processes. Comparative experimental results on the PHM2012 and XJTU-SY datasets demonstrate that TE-FEDformer outperforms other benchmark models. Ablation studies further validate that each module contributes positively to the overall performance, confirming the effectiveness of the proposed approach for RUL prediction.</p>
	]]></content:encoded>

	<dc:title>TE-FEDformer: A Time-Series-Enhanced FEDformer for Remaining Useful Life Prediction of Rolling Bearings</dc:title>
			<dc:creator>Yazhou Zhou</dc:creator>
			<dc:creator>Mingyang Tang</dc:creator>
			<dc:creator>Yunzhu Shan</dc:creator>
			<dc:creator>Wenbo Wang</dc:creator>
			<dc:creator>Man Zhou</dc:creator>
			<dc:creator>Yuchun Peng</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10080279</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-08-18</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-08-18</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>8</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>279</prism:startingPage>
		<prism:doi>10.3390/bdcc10080279</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/8/279</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/8/278">

	<title>BDCC, Vol. 10, Pages 278: Translate, Search, or Answer: Cost-Aware Cross-Lingual Retrieval for Kazakh Small Language Models</title>
	<link>https://www.mdpi.com/2504-2289/10/8/278</link>
	<description>Small Language Models (SLMs) enable efficient deployment, but their limited parameter count constrains factual knowledge, particularly in low-resource languages like Kazakh. Integrating live web search can address this limitation, though its effectiveness is difficult to measure due to sparse in-language web indices and answer leakage during benchmarking. In this study, we systematically compare zero-shot parametric generation, in-language retrieval, and cross-lingual (translate-then-retrieve) web search using three 4B-parameter SLMs in both reasoning and non-reasoning modes. To evaluate factuality without search-engine leakage, we introduce machine-translated Kazakh versions of the FreshQA and DefAn benchmarks, and use GPQA as a Google-proof adversarial control. We also assess robustness across three prompt complexities, from simple JSON constraints to adversarial warnings that instruct the model to treat potentially unreliable context with caution. Finally, we propose a training-free, self-aware router that uses majority voting over repeated self-verification decisions to determine when to answer parametrically, when to search the web, and when to escalate to a more capable cloud model. Our results show that cross-lingual retrieval substantially outperforms in-language search on global factuality tasks, nearly doubling accuracy on FreshQA, while direct in-language search remains preferable for localized cultural queries. The choice of retrieval language depends on the task and does not always favor English. Additionally, cross-lingual retrieval is not consistently superior, because the best option depends on where relevant information is indexed. Pareto analysis indicates that cross-lingual search is on or near the optimal accuracy&amp;amp;ndash;latency frontier, adding minimal overhead compared to direct search. The router identifies which query types warrant retrieval, and as a system it tracks or exceeds always-search accuracy while issuing fewer searches and approaching the always-cloud ceiling at a fraction of its cost; per-query discrimination within a task family is weaker, which we quantify explicitly. Overall, this work offers a framework for optimizing and accurately measuring cross-lingual RAG pipelines in low-resource settings.</description>
	<pubDate>2026-08-18</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 278: Translate, Search, or Answer: Cost-Aware Cross-Lingual Retrieval for Kazakh Small Language Models</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/8/278">doi: 10.3390/bdcc10080278</a></p>
	<p>Authors:
		Akylbek Maxutov
		Nūrali Medeu
		Vladimir Albrekht
		Danial Danenov
		Huseyin Atakan Varol
		</p>
	<p>Small Language Models (SLMs) enable efficient deployment, but their limited parameter count constrains factual knowledge, particularly in low-resource languages like Kazakh. Integrating live web search can address this limitation, though its effectiveness is difficult to measure due to sparse in-language web indices and answer leakage during benchmarking. In this study, we systematically compare zero-shot parametric generation, in-language retrieval, and cross-lingual (translate-then-retrieve) web search using three 4B-parameter SLMs in both reasoning and non-reasoning modes. To evaluate factuality without search-engine leakage, we introduce machine-translated Kazakh versions of the FreshQA and DefAn benchmarks, and use GPQA as a Google-proof adversarial control. We also assess robustness across three prompt complexities, from simple JSON constraints to adversarial warnings that instruct the model to treat potentially unreliable context with caution. Finally, we propose a training-free, self-aware router that uses majority voting over repeated self-verification decisions to determine when to answer parametrically, when to search the web, and when to escalate to a more capable cloud model. Our results show that cross-lingual retrieval substantially outperforms in-language search on global factuality tasks, nearly doubling accuracy on FreshQA, while direct in-language search remains preferable for localized cultural queries. The choice of retrieval language depends on the task and does not always favor English. Additionally, cross-lingual retrieval is not consistently superior, because the best option depends on where relevant information is indexed. Pareto analysis indicates that cross-lingual search is on or near the optimal accuracy&amp;amp;ndash;latency frontier, adding minimal overhead compared to direct search. The router identifies which query types warrant retrieval, and as a system it tracks or exceeds always-search accuracy while issuing fewer searches and approaching the always-cloud ceiling at a fraction of its cost; per-query discrimination within a task family is weaker, which we quantify explicitly. Overall, this work offers a framework for optimizing and accurately measuring cross-lingual RAG pipelines in low-resource settings.</p>
	]]></content:encoded>

	<dc:title>Translate, Search, or Answer: Cost-Aware Cross-Lingual Retrieval for Kazakh Small Language Models</dc:title>
			<dc:creator>Akylbek Maxutov</dc:creator>
			<dc:creator>Nūrali Medeu</dc:creator>
			<dc:creator>Vladimir Albrekht</dc:creator>
			<dc:creator>Danial Danenov</dc:creator>
			<dc:creator>Huseyin Atakan Varol</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10080278</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-08-18</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-08-18</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>8</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>278</prism:startingPage>
		<prism:doi>10.3390/bdcc10080278</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/8/278</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/8/277">

	<title>BDCC, Vol. 10, Pages 277: Sequential 2D&amp;ndash;3D Recognition for Privacy-Sensitive Object Extraction from 3D Point Clouds</title>
	<link>https://www.mdpi.com/2504-2289/10/8/277</link>
	<description>With the growing use of digital twins and 3D city models, 3D point cloud data have become increasingly important. Such data, however, may contain privacy-sensitive objects, including people and vehicles, which poses challenges for public release and secondary use. This study proposes a sequential 2D&amp;amp;ndash;3D recognition framework for extracting privacy-sensitive objects by integrating 2D image recognition and 3D point cloud recognition. The proposed framework first detects candidate regions in images and associates them with the corresponding 3D point cloud through multi-view projection, after which 3D semantic segmentation is applied only to the candidate point cloud. By restricting 3D recognition to candidate regions, the proposed method suppresses background-point contamination while reducing unnecessary 3D processing. We evaluate the method using SfM-derived 3D point clouds containing people and vehicles. The results show that the proposed method achieves higher F-scores than the selected direct 2D-projection and 3D-only baselines, reflecting a better balance between precision and recall. These findings suggest that sequentially combining 2D image recognition with 3D point cloud recognition provides an effective approach for privacy-sensitive object extraction and supports the privacy-preserving publication and secondary use of digital twins and 3D city models.</description>
	<pubDate>2026-08-18</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 277: Sequential 2D&amp;ndash;3D Recognition for Privacy-Sensitive Object Extraction from 3D Point Clouds</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/8/277">doi: 10.3390/bdcc10080277</a></p>
	<p>Authors:
		Yusuke Shinwashi
		Etsuji Kitagawa
		Satoshi Abiko
		Kennosuke Takada
		Ryo Kato
		</p>
	<p>With the growing use of digital twins and 3D city models, 3D point cloud data have become increasingly important. Such data, however, may contain privacy-sensitive objects, including people and vehicles, which poses challenges for public release and secondary use. This study proposes a sequential 2D&amp;amp;ndash;3D recognition framework for extracting privacy-sensitive objects by integrating 2D image recognition and 3D point cloud recognition. The proposed framework first detects candidate regions in images and associates them with the corresponding 3D point cloud through multi-view projection, after which 3D semantic segmentation is applied only to the candidate point cloud. By restricting 3D recognition to candidate regions, the proposed method suppresses background-point contamination while reducing unnecessary 3D processing. We evaluate the method using SfM-derived 3D point clouds containing people and vehicles. The results show that the proposed method achieves higher F-scores than the selected direct 2D-projection and 3D-only baselines, reflecting a better balance between precision and recall. These findings suggest that sequentially combining 2D image recognition with 3D point cloud recognition provides an effective approach for privacy-sensitive object extraction and supports the privacy-preserving publication and secondary use of digital twins and 3D city models.</p>
	]]></content:encoded>

	<dc:title>Sequential 2D&amp;amp;ndash;3D Recognition for Privacy-Sensitive Object Extraction from 3D Point Clouds</dc:title>
			<dc:creator>Yusuke Shinwashi</dc:creator>
			<dc:creator>Etsuji Kitagawa</dc:creator>
			<dc:creator>Satoshi Abiko</dc:creator>
			<dc:creator>Kennosuke Takada</dc:creator>
			<dc:creator>Ryo Kato</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10080277</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-08-18</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-08-18</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>8</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>277</prism:startingPage>
		<prism:doi>10.3390/bdcc10080277</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/8/277</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/8/276">

	<title>BDCC, Vol. 10, Pages 276: Accurate and Robust Multimodal Emotion Recognition for Human&amp;ndash;Robot Interaction via Dynamic Graph Learning with Pairwise Cross-Modal Alignment</title>
	<link>https://www.mdpi.com/2504-2289/10/8/276</link>
	<description>Multimodal emotion recognition in conversation (MERC) aims to identify the emotions in each utterance by modeling textual, acoustic, and visual evidence. Compared with unimodal emotion recognition in conversation (ERC), MERC can leverage complementary textual, acoustic, and visual information to support more accurate and consistent emotion inference. However, coordinating intramodal contextual dependencies, cross-modal alignment, and temporal affective dynamics within a unified framework in MERC is challenging. Existing solutions have advanced MERC through contextual modeling, multimodal fusion, and graph-based reasoning, but they still often rely on static relational assumptions or stage-wise coordination of modalities. This limits their ability to jointly model fine-grained relations, selective cross-modal interactions, and dynamic changes in emotion. To address these issues, we propose DGL-PCA (dynamic graph learning with pairwise cross-modal alignment), a dynamic graph-based framework for MERC. The model combines time-aware relation construction, dynamic time-aware heterogeneous graph modeling, and pairwise cross-modal alignment to improve prediction accuracy. This coordinates temporal affective dynamics, structured dialogue context, and multimodal interaction more explicitly than conventional coarse fusion or static graph formulations. Extensive experiments on IEMOCAP and CMU-MOSEI show that DGL-PCA consistently improves weighted F1 by 1.08&amp;amp;ndash;19.93% across all reproduced baselines. It achieves 70.02% and 83.91% weighted F1 on the IEMOCAP 6-way and 4-way settings, respectively, and 44.93% and 84.01% weighted F1 on the CMU-MOSEI 7-way and 2-way settings, respectively. Utterance-masking results demonstrated the robustness of the proposed method under dynamic emotional changes. In a 79-utterance IEMOCAP dialogue with 39 adjacent emotion transitions and up to nine transitions within a 15-utterance window, average weighted F1 slightly decreases by 1.87% in the case of any missing utterance, indicating long-context prediction stability under frequent emotion shifts. The proposed model paves the way for developing human-level emotion understanding capability of robots.</description>
	<pubDate>2026-08-18</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 276: Accurate and Robust Multimodal Emotion Recognition for Human&amp;ndash;Robot Interaction via Dynamic Graph Learning with Pairwise Cross-Modal Alignment</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/8/276">doi: 10.3390/bdcc10080276</a></p>
	<p>Authors:
		Xinyang Zhou
		Jiahao Wu
		Hongming Xu
		Jinghan Mei
		Zeyang Chen
		Junxiong Zhang
		Yitong Chen
		Yanrui Jin
		Chengliang Liu
		Chenggang Yuan
		</p>
	<p>Multimodal emotion recognition in conversation (MERC) aims to identify the emotions in each utterance by modeling textual, acoustic, and visual evidence. Compared with unimodal emotion recognition in conversation (ERC), MERC can leverage complementary textual, acoustic, and visual information to support more accurate and consistent emotion inference. However, coordinating intramodal contextual dependencies, cross-modal alignment, and temporal affective dynamics within a unified framework in MERC is challenging. Existing solutions have advanced MERC through contextual modeling, multimodal fusion, and graph-based reasoning, but they still often rely on static relational assumptions or stage-wise coordination of modalities. This limits their ability to jointly model fine-grained relations, selective cross-modal interactions, and dynamic changes in emotion. To address these issues, we propose DGL-PCA (dynamic graph learning with pairwise cross-modal alignment), a dynamic graph-based framework for MERC. The model combines time-aware relation construction, dynamic time-aware heterogeneous graph modeling, and pairwise cross-modal alignment to improve prediction accuracy. This coordinates temporal affective dynamics, structured dialogue context, and multimodal interaction more explicitly than conventional coarse fusion or static graph formulations. Extensive experiments on IEMOCAP and CMU-MOSEI show that DGL-PCA consistently improves weighted F1 by 1.08&amp;amp;ndash;19.93% across all reproduced baselines. It achieves 70.02% and 83.91% weighted F1 on the IEMOCAP 6-way and 4-way settings, respectively, and 44.93% and 84.01% weighted F1 on the CMU-MOSEI 7-way and 2-way settings, respectively. Utterance-masking results demonstrated the robustness of the proposed method under dynamic emotional changes. In a 79-utterance IEMOCAP dialogue with 39 adjacent emotion transitions and up to nine transitions within a 15-utterance window, average weighted F1 slightly decreases by 1.87% in the case of any missing utterance, indicating long-context prediction stability under frequent emotion shifts. The proposed model paves the way for developing human-level emotion understanding capability of robots.</p>
	]]></content:encoded>

	<dc:title>Accurate and Robust Multimodal Emotion Recognition for Human&amp;amp;ndash;Robot Interaction via Dynamic Graph Learning with Pairwise Cross-Modal Alignment</dc:title>
			<dc:creator>Xinyang Zhou</dc:creator>
			<dc:creator>Jiahao Wu</dc:creator>
			<dc:creator>Hongming Xu</dc:creator>
			<dc:creator>Jinghan Mei</dc:creator>
			<dc:creator>Zeyang Chen</dc:creator>
			<dc:creator>Junxiong Zhang</dc:creator>
			<dc:creator>Yitong Chen</dc:creator>
			<dc:creator>Yanrui Jin</dc:creator>
			<dc:creator>Chengliang Liu</dc:creator>
			<dc:creator>Chenggang Yuan</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10080276</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-08-18</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-08-18</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>8</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>276</prism:startingPage>
		<prism:doi>10.3390/bdcc10080276</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/8/276</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/8/275">

	<title>BDCC, Vol. 10, Pages 275: Declarative Causal Inference and Counterfactual Reasoning via SQL-Dialect Operators</title>
	<link>https://www.mdpi.com/2504-2289/10/8/275</link>
	<description>Relational databases power high-stakes decisions in lending, healthcare, and justice, yet SQL lacks native constructs for causal and counterfactual reasoning. Prior SQL-based causal systems address parts of this gap but do not unify treatment-effect estimation with counterfactual generation in a single, composable SQL surface. We present a system that extends the SQL dialect with two declarative operators: EXPLAIN_CAUSALLY_WHY (&amp;amp;psi;) for estimating average and conditional treatment effects via meta-learners, and EXPLAIN_COUNTERFACTUAL (&amp;amp;phi;) for generating diverse, constraint-respecting alternatives via a hybrid KD-tree/LSH pipeline. Both operators consume standard SQL relations (joins, filters, projections) and return table-valued results with optional diagnostics, confidence intervals, and feasibility metrics. We formalize the operators in relational algebra, and describe our prototype system called PsiQL. On four evaluation datasets, PsiQL recovers a protective TWINS treatment effect, returns a non-significant COMPAS point ATE with imbalance diagnostics, flags HMDA covariate imbalance via built-in SMD checks, and generates constraint-respecting counterfactuals; a synthetic Census run serves as a balanced pipeline proof-of-concept alongside a real ACS diagnostic under severe imbalance.</description>
	<pubDate>2026-08-17</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 275: Declarative Causal Inference and Counterfactual Reasoning via SQL-Dialect Operators</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/8/275">doi: 10.3390/bdcc10080275</a></p>
	<p>Authors:
		Ronnit Peter
		Suprio Ray
		Moulay A. Akhloufi
		</p>
	<p>Relational databases power high-stakes decisions in lending, healthcare, and justice, yet SQL lacks native constructs for causal and counterfactual reasoning. Prior SQL-based causal systems address parts of this gap but do not unify treatment-effect estimation with counterfactual generation in a single, composable SQL surface. We present a system that extends the SQL dialect with two declarative operators: EXPLAIN_CAUSALLY_WHY (&amp;amp;psi;) for estimating average and conditional treatment effects via meta-learners, and EXPLAIN_COUNTERFACTUAL (&amp;amp;phi;) for generating diverse, constraint-respecting alternatives via a hybrid KD-tree/LSH pipeline. Both operators consume standard SQL relations (joins, filters, projections) and return table-valued results with optional diagnostics, confidence intervals, and feasibility metrics. We formalize the operators in relational algebra, and describe our prototype system called PsiQL. On four evaluation datasets, PsiQL recovers a protective TWINS treatment effect, returns a non-significant COMPAS point ATE with imbalance diagnostics, flags HMDA covariate imbalance via built-in SMD checks, and generates constraint-respecting counterfactuals; a synthetic Census run serves as a balanced pipeline proof-of-concept alongside a real ACS diagnostic under severe imbalance.</p>
	]]></content:encoded>

	<dc:title>Declarative Causal Inference and Counterfactual Reasoning via SQL-Dialect Operators</dc:title>
			<dc:creator>Ronnit Peter</dc:creator>
			<dc:creator>Suprio Ray</dc:creator>
			<dc:creator>Moulay A. Akhloufi</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10080275</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-08-17</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-08-17</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>8</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>275</prism:startingPage>
		<prism:doi>10.3390/bdcc10080275</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/8/275</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/8/274">

	<title>BDCC, Vol. 10, Pages 274: Benchmarking Supervised Classifiers for Concurrent Multidomain Dropout-Intention Attributions in Higher Education: Evidence from a Colombian Public University</title>
	<link>https://www.mdpi.com/2504-2289/10/8/274</link>
	<description>Student retention analytics often treats withdrawal as a single outcome, although students may attribute dropout intention to personal, socioeconomic, and academic pressures simultaneously. We benchmarked nine supervised classifiers for identifying a concurrent three-domain attribution profile in a cross-sectional survey of 333 undergraduates at a Colombian public university campus. The response came from a semi-structured weight-allocation item; an audit found that literal label matching altered 21 classifications because of spelling variants and decimal notation. Nine classifiers&amp;amp;mdash;logistic regression, decision tree, random forest, neural network, Gaussian Na&amp;amp;iuml;ve Bayes, k-nearest neighbors, AdaBoost, gradient boosting, and XGBoost&amp;amp;mdash;were fitted using six pre-specified predictors. Models were compared by repeated nested stratified cross-validation (five outer folds, three repeats), inner tuning, fold-contained preprocessing, and training-only threshold selection. The concurrent profile occurred in 256 students (76.9%). Logistic regression achieved the highest mean held-out ROC AUC (0.674, 95% CI 0.644&amp;amp;ndash;0.704), closely followed by random forest (0.671, 0.641&amp;amp;ndash;0.702); their paired difference was nonsignificant after Holm adjustment. Logistic regression had the highest F1 score (0.788), whereas random forest had the highest balanced accuracy (0.617). AdaBoost did not retain its apparent single-holdout advantage. Housing and financial aid had the largest held-out permutation importance. The predictors provided moderate discrimination of a perceptual profile, not a validated prediction of future dropout. Outcome auditing and leakage-free validation materially changed the model ranking.</description>
	<pubDate>2026-08-16</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 274: Benchmarking Supervised Classifiers for Concurrent Multidomain Dropout-Intention Attributions in Higher Education: Evidence from a Colombian Public University</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/8/274">doi: 10.3390/bdcc10080274</a></p>
	<p>Authors:
		Marieth Agnes Guillen-García
		Osnamir Elias Bru-Cordero
		Cristian David Correa-Álvarez
		</p>
	<p>Student retention analytics often treats withdrawal as a single outcome, although students may attribute dropout intention to personal, socioeconomic, and academic pressures simultaneously. We benchmarked nine supervised classifiers for identifying a concurrent three-domain attribution profile in a cross-sectional survey of 333 undergraduates at a Colombian public university campus. The response came from a semi-structured weight-allocation item; an audit found that literal label matching altered 21 classifications because of spelling variants and decimal notation. Nine classifiers&amp;amp;mdash;logistic regression, decision tree, random forest, neural network, Gaussian Na&amp;amp;iuml;ve Bayes, k-nearest neighbors, AdaBoost, gradient boosting, and XGBoost&amp;amp;mdash;were fitted using six pre-specified predictors. Models were compared by repeated nested stratified cross-validation (five outer folds, three repeats), inner tuning, fold-contained preprocessing, and training-only threshold selection. The concurrent profile occurred in 256 students (76.9%). Logistic regression achieved the highest mean held-out ROC AUC (0.674, 95% CI 0.644&amp;amp;ndash;0.704), closely followed by random forest (0.671, 0.641&amp;amp;ndash;0.702); their paired difference was nonsignificant after Holm adjustment. Logistic regression had the highest F1 score (0.788), whereas random forest had the highest balanced accuracy (0.617). AdaBoost did not retain its apparent single-holdout advantage. Housing and financial aid had the largest held-out permutation importance. The predictors provided moderate discrimination of a perceptual profile, not a validated prediction of future dropout. Outcome auditing and leakage-free validation materially changed the model ranking.</p>
	]]></content:encoded>

	<dc:title>Benchmarking Supervised Classifiers for Concurrent Multidomain Dropout-Intention Attributions in Higher Education: Evidence from a Colombian Public University</dc:title>
			<dc:creator>Marieth Agnes Guillen-García</dc:creator>
			<dc:creator>Osnamir Elias Bru-Cordero</dc:creator>
			<dc:creator>Cristian David Correa-Álvarez</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10080274</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-08-16</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-08-16</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>8</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>274</prism:startingPage>
		<prism:doi>10.3390/bdcc10080274</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/8/274</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/8/273">

	<title>BDCC, Vol. 10, Pages 273: Deep Learning in Farming: A Systematic Evidence-Weighted Review of Applications, Validation Gaps, and Emerging Frontiers</title>
	<link>https://www.mdpi.com/2504-2289/10/8/273</link>
	<description>This article presents a systematic, evidence-weighted review of Deep Learning (DL) in farming, with a primary emphasis on the agricultural production stage and on four operational domains: precision crop management, precision livestock farming, soil and water resource management, and autonomous agricultural systems. Following a PRISMA 2020-oriented protocol, 56 primary studies (42 with quantitative results) were retained from an initial pool of 18,731 records. The reviewed literature reports applications in plant disease detection, weed recognition, yield prediction, fruit detection, livestock identification and health monitoring, soil-property estimation, crop-water-stress assessment, and robotic perception. High performance on controlled datasets, however, is frequently reported without external, temporal, or cross-site validation, making practical generalisation difficult to establish: in one widely cited benchmark, disease-classification accuracy fell from above 99% on held-out laboratory images to 31.4% on field-acquired images of the same classes. Persistent weaknesses include the limited availability of public benchmarks, inconsistent validation protocols, limited interpretability, fragmented data governance, and insufficient techno-economic analysis. The review argues that the next stage of agricultural AI should be judged less by isolated benchmark scores and more by field realism, reproducibility, deployment maturity, and practical usefulness. Unlike broad surveys that mainly catalogue architectures and applications, this review interprets the literature according to dataset representativeness, validation protocols, benchmarking transparency, deployment realism, and reproducibility, distinguishing algorithmic performance under controlled conditions from practical readiness for real farming environments. The most promising research directions include self-supervised and multimodal learning, explainable and privacy-preserving AI, edge-aware deployment, and hybrid process-informed models.</description>
	<pubDate>2026-08-14</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 273: Deep Learning in Farming: A Systematic Evidence-Weighted Review of Applications, Validation Gaps, and Emerging Frontiers</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/8/273">doi: 10.3390/bdcc10080273</a></p>
	<p>Authors:
		Vito Domenico Amodio
		Lerina Aversano
		Vincenzo Dentamaro
		Felice Franchini
		</p>
	<p>This article presents a systematic, evidence-weighted review of Deep Learning (DL) in farming, with a primary emphasis on the agricultural production stage and on four operational domains: precision crop management, precision livestock farming, soil and water resource management, and autonomous agricultural systems. Following a PRISMA 2020-oriented protocol, 56 primary studies (42 with quantitative results) were retained from an initial pool of 18,731 records. The reviewed literature reports applications in plant disease detection, weed recognition, yield prediction, fruit detection, livestock identification and health monitoring, soil-property estimation, crop-water-stress assessment, and robotic perception. High performance on controlled datasets, however, is frequently reported without external, temporal, or cross-site validation, making practical generalisation difficult to establish: in one widely cited benchmark, disease-classification accuracy fell from above 99% on held-out laboratory images to 31.4% on field-acquired images of the same classes. Persistent weaknesses include the limited availability of public benchmarks, inconsistent validation protocols, limited interpretability, fragmented data governance, and insufficient techno-economic analysis. The review argues that the next stage of agricultural AI should be judged less by isolated benchmark scores and more by field realism, reproducibility, deployment maturity, and practical usefulness. Unlike broad surveys that mainly catalogue architectures and applications, this review interprets the literature according to dataset representativeness, validation protocols, benchmarking transparency, deployment realism, and reproducibility, distinguishing algorithmic performance under controlled conditions from practical readiness for real farming environments. The most promising research directions include self-supervised and multimodal learning, explainable and privacy-preserving AI, edge-aware deployment, and hybrid process-informed models.</p>
	]]></content:encoded>

	<dc:title>Deep Learning in Farming: A Systematic Evidence-Weighted Review of Applications, Validation Gaps, and Emerging Frontiers</dc:title>
			<dc:creator>Vito Domenico Amodio</dc:creator>
			<dc:creator>Lerina Aversano</dc:creator>
			<dc:creator>Vincenzo Dentamaro</dc:creator>
			<dc:creator>Felice Franchini</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10080273</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-08-14</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-08-14</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>8</prism:number>
	<prism:section>Systematic Review</prism:section>
	<prism:startingPage>273</prism:startingPage>
		<prism:doi>10.3390/bdcc10080273</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/8/273</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/8/272">

	<title>BDCC, Vol. 10, Pages 272: Layerwise Conditioned Backpropagation: A Curvature-Aware Reparameterization of the Backward Pass with Convergence Guarantees</title>
	<link>https://www.mdpi.com/2504-2289/10/8/272</link>
	<description>Backpropagation is less a single algorithm than a pipeline of choices: how the error signal is propagated, how the weight gradient is assembled, and how the update is applied. This paper revisits three consecutive steps and proposes small, mathematically transparent modifications that improve gradient scaling and conditioning without changing the represented function class. The resulting method, Conditioned Backpropagation(CBP), combines (i) a layerwise gradient-norm equalization that counters the geometric depth dependence of the backpropagated error; (ii) an activation-centering reparameterization that removes the dominant rank-one mean term from the per-layer curvature; and (iii) a damped diagonal preconditioner that is positive-definite by construction. The composite operator is a bounded positive-definite preconditioner, so the method inherits standard nonconvex, Polyak&amp;amp;ndash;&amp;amp;#321;ojasiewicz, and stochastic convergence guarantees at the per-step cost of ordinary backpropagation. No prior method composes these three repairs into one operator with a joint boundedness and positive-definiteness guarantee. Two further results, both new, concern equalization. On a block-structured strongly convex model, and for the curvature-equalizing target that the implemented gradient-energy equalizer approximates up to a quantified heterogeneity factor, equalization makes the convergence rate depth-uniform; the bounded-clip version that is actually run stays depth-uniform up to a clip-determined depth and retains a constant-factor improvement beyond it. Controlled experiments, run over ten or more seeds with paired significance tests, confirm the mechanisms: Equalization compresses an order-of-magnitude per-layer gradient disparity, centering cuts the top curvature eigenvalue about threefold and yields the lowest training loss, and the configurations combining centering with the damped preconditioner, including the full method, converge fastest. The effects persist on MNIST and on CIFAR-10 with a small residual convolutional network, at a measured per-iteration overhead below about twice that of Adam. Generalization is comparable across methods, and no end-to-end depth-scaling advantage is claimed, keeping the contribution focused on optimization geometry.</description>
	<pubDate>2026-08-13</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 272: Layerwise Conditioned Backpropagation: A Curvature-Aware Reparameterization of the Backward Pass with Convergence Guarantees</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/8/272">doi: 10.3390/bdcc10080272</a></p>
	<p>Authors:
		Maikel Leon
		</p>
	<p>Backpropagation is less a single algorithm than a pipeline of choices: how the error signal is propagated, how the weight gradient is assembled, and how the update is applied. This paper revisits three consecutive steps and proposes small, mathematically transparent modifications that improve gradient scaling and conditioning without changing the represented function class. The resulting method, Conditioned Backpropagation(CBP), combines (i) a layerwise gradient-norm equalization that counters the geometric depth dependence of the backpropagated error; (ii) an activation-centering reparameterization that removes the dominant rank-one mean term from the per-layer curvature; and (iii) a damped diagonal preconditioner that is positive-definite by construction. The composite operator is a bounded positive-definite preconditioner, so the method inherits standard nonconvex, Polyak&amp;amp;ndash;&amp;amp;#321;ojasiewicz, and stochastic convergence guarantees at the per-step cost of ordinary backpropagation. No prior method composes these three repairs into one operator with a joint boundedness and positive-definiteness guarantee. Two further results, both new, concern equalization. On a block-structured strongly convex model, and for the curvature-equalizing target that the implemented gradient-energy equalizer approximates up to a quantified heterogeneity factor, equalization makes the convergence rate depth-uniform; the bounded-clip version that is actually run stays depth-uniform up to a clip-determined depth and retains a constant-factor improvement beyond it. Controlled experiments, run over ten or more seeds with paired significance tests, confirm the mechanisms: Equalization compresses an order-of-magnitude per-layer gradient disparity, centering cuts the top curvature eigenvalue about threefold and yields the lowest training loss, and the configurations combining centering with the damped preconditioner, including the full method, converge fastest. The effects persist on MNIST and on CIFAR-10 with a small residual convolutional network, at a measured per-iteration overhead below about twice that of Adam. Generalization is comparable across methods, and no end-to-end depth-scaling advantage is claimed, keeping the contribution focused on optimization geometry.</p>
	]]></content:encoded>

	<dc:title>Layerwise Conditioned Backpropagation: A Curvature-Aware Reparameterization of the Backward Pass with Convergence Guarantees</dc:title>
			<dc:creator>Maikel Leon</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10080272</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-08-13</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-08-13</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>8</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>272</prism:startingPage>
		<prism:doi>10.3390/bdcc10080272</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/8/272</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/8/271">

	<title>BDCC, Vol. 10, Pages 271: Enhanced TabNet with Entmax-Based Sparse Attention and Modified GLU for Interpretable Cardiovascular Risk Prediction Using an Edge-IoT Framework</title>
	<link>https://www.mdpi.com/2504-2289/10/8/271</link>
	<description>Cardiovascular diseases remain the leading global cause of mortality, necessitating continuous monitoring solutions that extend beyond clinical settings. This paper proposes a real-time, end-to-end Edge-IoT framework for cardiovascular risk assessment that integrates biomedical signal acquisition, edge processing, and interpretable deep learning. The system includes a three-tier architecture: (i) physiological signal acquisition using AD8232 ECG, MAX30102 photoplethysmography, DS18B20 temperature, and NEO-6M GPS sensors interfaced with an ESP32 microcontroller; (ii) real-time signal preprocessing, including digital filtering, normalisation, and PQRST feature extraction performed at the edge; and (iii) cloud-based analytics using an Enhanced TabNet classifier with modified attention mechanisms for cardiovascular risk prediction. The Enhanced TabNet architecture incorporates Entmax-based sparse attention and modified Gated Linear Units to improve predictive performance and clinical interpretability. Signal quality enhancement using Kalman filtering and class imbalance correction using SMOTE further support robust model performance. The Enhanced TabNet model achieves 97.43% accuracy, 96.18% precision, and 97.24% recall on the combined Cleveland, Hungarian, Switzerland, Long Beach VA, and Statlog heart disease datasets (n=1190). The developed Edge-IoT prototype maintains an end-to-end communication and processing latency below 200 ms. The framework also includes automated risk alert generation via SMS when the predicted cardiovascular risk probability exceeds a predefined threshold (e.g., 0.85), including the patient&amp;amp;rsquo;s vital information and geolocation to support emergency response. The integrated edge-cloud architecture with attention-based feature selection provides interpretable cardiovascular risk predictions while maintaining the computational efficiency required for potential continuous patient monitoring outside hospital settings.</description>
	<pubDate>2026-08-11</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 271: Enhanced TabNet with Entmax-Based Sparse Attention and Modified GLU for Interpretable Cardiovascular Risk Prediction Using an Edge-IoT Framework</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/8/271">doi: 10.3390/bdcc10080271</a></p>
	<p>Authors:
		Mehboob Zahedi
		Dokhyl AlQahtani
		Bader Alhasson
		Emad A. Mohamed
		Pradeep Kumar Dabla
		Shyamalendu Kandar
		</p>
	<p>Cardiovascular diseases remain the leading global cause of mortality, necessitating continuous monitoring solutions that extend beyond clinical settings. This paper proposes a real-time, end-to-end Edge-IoT framework for cardiovascular risk assessment that integrates biomedical signal acquisition, edge processing, and interpretable deep learning. The system includes a three-tier architecture: (i) physiological signal acquisition using AD8232 ECG, MAX30102 photoplethysmography, DS18B20 temperature, and NEO-6M GPS sensors interfaced with an ESP32 microcontroller; (ii) real-time signal preprocessing, including digital filtering, normalisation, and PQRST feature extraction performed at the edge; and (iii) cloud-based analytics using an Enhanced TabNet classifier with modified attention mechanisms for cardiovascular risk prediction. The Enhanced TabNet architecture incorporates Entmax-based sparse attention and modified Gated Linear Units to improve predictive performance and clinical interpretability. Signal quality enhancement using Kalman filtering and class imbalance correction using SMOTE further support robust model performance. The Enhanced TabNet model achieves 97.43% accuracy, 96.18% precision, and 97.24% recall on the combined Cleveland, Hungarian, Switzerland, Long Beach VA, and Statlog heart disease datasets (n=1190). The developed Edge-IoT prototype maintains an end-to-end communication and processing latency below 200 ms. The framework also includes automated risk alert generation via SMS when the predicted cardiovascular risk probability exceeds a predefined threshold (e.g., 0.85), including the patient&amp;amp;rsquo;s vital information and geolocation to support emergency response. The integrated edge-cloud architecture with attention-based feature selection provides interpretable cardiovascular risk predictions while maintaining the computational efficiency required for potential continuous patient monitoring outside hospital settings.</p>
	]]></content:encoded>

	<dc:title>Enhanced TabNet with Entmax-Based Sparse Attention and Modified GLU for Interpretable Cardiovascular Risk Prediction Using an Edge-IoT Framework</dc:title>
			<dc:creator>Mehboob Zahedi</dc:creator>
			<dc:creator>Dokhyl AlQahtani</dc:creator>
			<dc:creator>Bader Alhasson</dc:creator>
			<dc:creator>Emad A. Mohamed</dc:creator>
			<dc:creator>Pradeep Kumar Dabla</dc:creator>
			<dc:creator>Shyamalendu Kandar</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10080271</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-08-11</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-08-11</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>8</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>271</prism:startingPage>
		<prism:doi>10.3390/bdcc10080271</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/8/271</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/8/270">

	<title>BDCC, Vol. 10, Pages 270: Evaluating Machine Learning and Deep Learning Models for Early Detection of Alzheimer&amp;rsquo;s and Parkinson&amp;rsquo;s Disease: An Explainable Dual-Dataset Study on Clinical Generalizability</title>
	<link>https://www.mdpi.com/2504-2289/10/8/270</link>
	<description>Neurodegenerative diseases such as Alzheimer&amp;amp;rsquo;s disease (AD) and Parkinson&amp;amp;rsquo;s disease (PD) pose a significant global healthcare burden due to challenges in early diagnosis. This study investigates the clinical generalizability of machine learning (ML) and deep learning (DL) models for early AD and PD classification across multiple data modalities. A dual-dataset framework was employed, combining global benchmarks (ADNI, PPMI, OASIS, and UCI voice) with local clinical data from King Fahd Hospital of the University (KFHU) in Saudi Arabia. We evaluated ensemble methods, SVMs, neural networks, CNNs, and LSTMs. On structured global data, tree-based ensembles achieved the best performance, with Random Forest reaching 88.24% accuracy for PD and Gradient Boosting achieving 94.44% for AD. For neuroimaging, an LSTM on CNN features attained 98.68% accuracy on a curated MRI dataset. A critical finding was a substantial generalization gap: models excelling on global data showed markedly reduced performance on local KFHU data, with AUC values between 0.50 and 0.77. This degradation is attributed to real-world clinical challenges including severe class imbalance, diagnostic uncertainty in EHRs, and heterogeneous feature representations. The results underscore that data quality and modality are often more consequential than algorithmic complexity. This study provides a reproducible validation framework, highlights the necessity of institution-specific evaluation, establishes performance benchmarks for Saudi healthcare (with the caveat that local sample sizes remain small and the results should be interpreted as exploratory), and demonstrates the use of explainable AI to validate clinical relevance.</description>
	<pubDate>2026-08-11</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 270: Evaluating Machine Learning and Deep Learning Models for Early Detection of Alzheimer&amp;rsquo;s and Parkinson&amp;rsquo;s Disease: An Explainable Dual-Dataset Study on Clinical Generalizability</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/8/270">doi: 10.3390/bdcc10080270</a></p>
	<p>Authors:
		Muneera Mohammed Al-Dossary
		Atta Rahman
		</p>
	<p>Neurodegenerative diseases such as Alzheimer&amp;amp;rsquo;s disease (AD) and Parkinson&amp;amp;rsquo;s disease (PD) pose a significant global healthcare burden due to challenges in early diagnosis. This study investigates the clinical generalizability of machine learning (ML) and deep learning (DL) models for early AD and PD classification across multiple data modalities. A dual-dataset framework was employed, combining global benchmarks (ADNI, PPMI, OASIS, and UCI voice) with local clinical data from King Fahd Hospital of the University (KFHU) in Saudi Arabia. We evaluated ensemble methods, SVMs, neural networks, CNNs, and LSTMs. On structured global data, tree-based ensembles achieved the best performance, with Random Forest reaching 88.24% accuracy for PD and Gradient Boosting achieving 94.44% for AD. For neuroimaging, an LSTM on CNN features attained 98.68% accuracy on a curated MRI dataset. A critical finding was a substantial generalization gap: models excelling on global data showed markedly reduced performance on local KFHU data, with AUC values between 0.50 and 0.77. This degradation is attributed to real-world clinical challenges including severe class imbalance, diagnostic uncertainty in EHRs, and heterogeneous feature representations. The results underscore that data quality and modality are often more consequential than algorithmic complexity. This study provides a reproducible validation framework, highlights the necessity of institution-specific evaluation, establishes performance benchmarks for Saudi healthcare (with the caveat that local sample sizes remain small and the results should be interpreted as exploratory), and demonstrates the use of explainable AI to validate clinical relevance.</p>
	]]></content:encoded>

	<dc:title>Evaluating Machine Learning and Deep Learning Models for Early Detection of Alzheimer&amp;amp;rsquo;s and Parkinson&amp;amp;rsquo;s Disease: An Explainable Dual-Dataset Study on Clinical Generalizability</dc:title>
			<dc:creator>Muneera Mohammed Al-Dossary</dc:creator>
			<dc:creator>Atta Rahman</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10080270</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-08-11</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-08-11</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>8</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>270</prism:startingPage>
		<prism:doi>10.3390/bdcc10080270</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/8/270</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/8/269">

	<title>BDCC, Vol. 10, Pages 269: From Unstructured Reports to Exploratory Causal Modeling: A Modality-Aware AI Pipeline for Infrastructure Delay Analysis</title>
	<link>https://www.mdpi.com/2504-2289/10/8/269</link>
	<description>Infrastructure project reports contain rich narrative evidence on delay causes, yet transforming such unstructured text into reliable causal knowledge remains challenging because reports mix confirmed events with hypothetical, conditional, or localized statements. This study proposes an eight-stage computational pipeline that converts infrastructure project evaluation reports into a Bayesian-network model for exploratory structure learning and probabilistic dependency modeling. The central methodological contribution is a modality-aware extraction layer that distinguishes confirmed, project-wide delay evidence from conditional, hypothetical, or component-level statements before causal analysis. The pipeline was evaluated on 55 road infrastructure project reports financed by the Asian Development Bank, the African Development Bank, and JICA, from which delay events across 15 cause categories were extracted and stratified by epistemic modality and scope. Ablation analysis shows that the principal dependency structure recovered by the Bayesian network is not recoverable without modality-aware filtering, indicating that evidence-quality stratification materially shapes downstream causal-structure exploration. Among the recovered dependencies, a financial-to-project-management pathway was the most consistent signal: its undirected skeleton edge was the only relationship recovered by all four causal-discovery algorithms tested (with the orientation determined only by the score-based search), its association was nominally positive&amp;amp;mdash;though weak and not uniformly discernible&amp;amp;mdash;across nine extraction models spanning three commercial vendors and open-weight families, and it is consistent with prior delay-factor literature. Its model-based scenario contrast (&amp;amp;Delta;P=+0.638, 95% CI [0.470,0.764]) is reported as hypothesis-generating rather than as a validated policy effect: under structure-learning uncertainty, the interval extends to zero, and the effect magnitude and the specific learned edge depend on the extraction model and the small effective sample. These findings suggest that incorporating modality awareness into narrative-evidence extraction improves the reliability of exploratory causal-structure analysis from infrastructure project reports.</description>
	<pubDate>2026-08-11</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 269: From Unstructured Reports to Exploratory Causal Modeling: A Modality-Aware AI Pipeline for Infrastructure Delay Analysis</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/8/269">doi: 10.3390/bdcc10080269</a></p>
	<p>Authors:
		Florence Gundidza
		Masato Kikuchi
		Tadachika Ozono
		</p>
	<p>Infrastructure project reports contain rich narrative evidence on delay causes, yet transforming such unstructured text into reliable causal knowledge remains challenging because reports mix confirmed events with hypothetical, conditional, or localized statements. This study proposes an eight-stage computational pipeline that converts infrastructure project evaluation reports into a Bayesian-network model for exploratory structure learning and probabilistic dependency modeling. The central methodological contribution is a modality-aware extraction layer that distinguishes confirmed, project-wide delay evidence from conditional, hypothetical, or component-level statements before causal analysis. The pipeline was evaluated on 55 road infrastructure project reports financed by the Asian Development Bank, the African Development Bank, and JICA, from which delay events across 15 cause categories were extracted and stratified by epistemic modality and scope. Ablation analysis shows that the principal dependency structure recovered by the Bayesian network is not recoverable without modality-aware filtering, indicating that evidence-quality stratification materially shapes downstream causal-structure exploration. Among the recovered dependencies, a financial-to-project-management pathway was the most consistent signal: its undirected skeleton edge was the only relationship recovered by all four causal-discovery algorithms tested (with the orientation determined only by the score-based search), its association was nominally positive&amp;amp;mdash;though weak and not uniformly discernible&amp;amp;mdash;across nine extraction models spanning three commercial vendors and open-weight families, and it is consistent with prior delay-factor literature. Its model-based scenario contrast (&amp;amp;Delta;P=+0.638, 95% CI [0.470,0.764]) is reported as hypothesis-generating rather than as a validated policy effect: under structure-learning uncertainty, the interval extends to zero, and the effect magnitude and the specific learned edge depend on the extraction model and the small effective sample. These findings suggest that incorporating modality awareness into narrative-evidence extraction improves the reliability of exploratory causal-structure analysis from infrastructure project reports.</p>
	]]></content:encoded>

	<dc:title>From Unstructured Reports to Exploratory Causal Modeling: A Modality-Aware AI Pipeline for Infrastructure Delay Analysis</dc:title>
			<dc:creator>Florence Gundidza</dc:creator>
			<dc:creator>Masato Kikuchi</dc:creator>
			<dc:creator>Tadachika Ozono</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10080269</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-08-11</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-08-11</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>8</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>269</prism:startingPage>
		<prism:doi>10.3390/bdcc10080269</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/8/269</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/8/268">

	<title>BDCC, Vol. 10, Pages 268: A Review of Human-AI Complementarities Across Multiple Dimensions of Organisational Complexity</title>
	<link>https://www.mdpi.com/2504-2289/10/8/268</link>
	<description>The growing capabilities of artificial intelligence (AI) have not translated straightforwardly into organisational value. A persistent disconnect&amp;amp;mdash;the &amp;amp;ldquo;last-mile problem&amp;amp;rdquo;&amp;amp;mdash;arises from structural gaps between idealised AI tasks and real-world organisational contexts. Synthesising insights from organisational theory, cognitive science, and computer science, we have developed a five-dimensional diagnostic framework that maps the challenges of human-AI collaboration across Integration, Representation, Scale, Temporality, and Adequacy gaps. These gaps illuminate how socio-technical complexity, contextualised problem representations, interdependencies among agents, dynamic environments, and limitations in current AI reasoning collectively constrain full automation and demand human judgement. By reviewing the historical evolution of AI&amp;amp;mdash;from symbolic systems to machine learning, generative models, and emerging agentic approaches&amp;amp;mdash;we show that augmentation remains the dominant and most viable mode of use in complex environments. An illustrative system-dynamics example demonstrates how improvements in algorithmic performance do not automatically yield proportional system-level gains. Overall, our framework provides researchers with a conceptual lens and practitioners with a diagnostic tool for assessing complementarities and informing the design of human-AI collaborations. The framework is offered as a conceptual synthesis and diagnostic instrument rather than an empirically validated model.</description>
	<pubDate>2026-08-11</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 268: A Review of Human-AI Complementarities Across Multiple Dimensions of Organisational Complexity</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/8/268">doi: 10.3390/bdcc10080268</a></p>
	<p>Authors:
		Ganesh Sankaran
		Marco A. Palomino
		Guido Siestrup
		</p>
	<p>The growing capabilities of artificial intelligence (AI) have not translated straightforwardly into organisational value. A persistent disconnect&amp;amp;mdash;the &amp;amp;ldquo;last-mile problem&amp;amp;rdquo;&amp;amp;mdash;arises from structural gaps between idealised AI tasks and real-world organisational contexts. Synthesising insights from organisational theory, cognitive science, and computer science, we have developed a five-dimensional diagnostic framework that maps the challenges of human-AI collaboration across Integration, Representation, Scale, Temporality, and Adequacy gaps. These gaps illuminate how socio-technical complexity, contextualised problem representations, interdependencies among agents, dynamic environments, and limitations in current AI reasoning collectively constrain full automation and demand human judgement. By reviewing the historical evolution of AI&amp;amp;mdash;from symbolic systems to machine learning, generative models, and emerging agentic approaches&amp;amp;mdash;we show that augmentation remains the dominant and most viable mode of use in complex environments. An illustrative system-dynamics example demonstrates how improvements in algorithmic performance do not automatically yield proportional system-level gains. Overall, our framework provides researchers with a conceptual lens and practitioners with a diagnostic tool for assessing complementarities and informing the design of human-AI collaborations. The framework is offered as a conceptual synthesis and diagnostic instrument rather than an empirically validated model.</p>
	]]></content:encoded>

	<dc:title>A Review of Human-AI Complementarities Across Multiple Dimensions of Organisational Complexity</dc:title>
			<dc:creator>Ganesh Sankaran</dc:creator>
			<dc:creator>Marco A. Palomino</dc:creator>
			<dc:creator>Guido Siestrup</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10080268</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-08-11</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-08-11</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>8</prism:number>
	<prism:section>Review</prism:section>
	<prism:startingPage>268</prism:startingPage>
		<prism:doi>10.3390/bdcc10080268</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/8/268</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/8/267">

	<title>BDCC, Vol. 10, Pages 267: A Self-Adaptive Agentic Mixture-of-Agent Families with Agent-to-Agent Communication for Automatic Sentiment Analysis</title>
	<link>https://www.mdpi.com/2504-2289/10/8/267</link>
	<description>Automatic sentiment analysis requires models that can effectively handle texts of varying levels of complexity. Monolithic methods use the same algorithm uniformly, without taking into account the intrinsic complexity of the input text. In response to this need, we developed a Dynamic Agentic Mixture-of-Agents with Inter-Agent Communication and Adaptive Routing for Robust Sentiment Analysis (DAMA-Sent). This approach merges three distinct algorithmic paradigms: statistical learning, deep learning, and attention models. The decomposition process is carried out in a sophisticated system featuring a hierarchical routing system and an inter-agent communication system based on differentiable attention. Furthermore, each agent has a self-reflection module, a weighting mechanism that takes uncertainty into account and allows the agent to assess its own reliability. Finally, an adaptive early exit system halts processing as soon as an appropriate confidence threshold is reached or the computation budget is exhausted. In-depth analyses conducted on a corpus of tweets from American airlines reveal that the suggested approach can adjust to the intrinsic variability of textual complexity and surpasses static ensemble methods in terms of accuracy and computational cost, achieving an accuracy of 95.96%. Additional validations corroborate these trends, demonstrating both the structural relevance and the validity of our proposed framework.</description>
	<pubDate>2026-08-10</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 267: A Self-Adaptive Agentic Mixture-of-Agent Families with Agent-to-Agent Communication for Automatic Sentiment Analysis</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/8/267">doi: 10.3390/bdcc10080267</a></p>
	<p>Authors:
		Wiam Saidi
		Boutaina Satouri
		Abdellatif El Abderrahmani
		Khalid Satori
		</p>
	<p>Automatic sentiment analysis requires models that can effectively handle texts of varying levels of complexity. Monolithic methods use the same algorithm uniformly, without taking into account the intrinsic complexity of the input text. In response to this need, we developed a Dynamic Agentic Mixture-of-Agents with Inter-Agent Communication and Adaptive Routing for Robust Sentiment Analysis (DAMA-Sent). This approach merges three distinct algorithmic paradigms: statistical learning, deep learning, and attention models. The decomposition process is carried out in a sophisticated system featuring a hierarchical routing system and an inter-agent communication system based on differentiable attention. Furthermore, each agent has a self-reflection module, a weighting mechanism that takes uncertainty into account and allows the agent to assess its own reliability. Finally, an adaptive early exit system halts processing as soon as an appropriate confidence threshold is reached or the computation budget is exhausted. In-depth analyses conducted on a corpus of tweets from American airlines reveal that the suggested approach can adjust to the intrinsic variability of textual complexity and surpasses static ensemble methods in terms of accuracy and computational cost, achieving an accuracy of 95.96%. Additional validations corroborate these trends, demonstrating both the structural relevance and the validity of our proposed framework.</p>
	]]></content:encoded>

	<dc:title>A Self-Adaptive Agentic Mixture-of-Agent Families with Agent-to-Agent Communication for Automatic Sentiment Analysis</dc:title>
			<dc:creator>Wiam Saidi</dc:creator>
			<dc:creator>Boutaina Satouri</dc:creator>
			<dc:creator>Abdellatif El Abderrahmani</dc:creator>
			<dc:creator>Khalid Satori</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10080267</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-08-10</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-08-10</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>8</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>267</prism:startingPage>
		<prism:doi>10.3390/bdcc10080267</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/8/267</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/8/266">

	<title>BDCC, Vol. 10, Pages 266: Descriptive Process Mining of Pulmonary Clinical Pathways Before and During COVID-19</title>
	<link>https://www.mdpi.com/2504-2289/10/8/266</link>
	<description>Understanding how clinical pathways evolve over time is essential for characterizing care processes. It also helps identify potential shifts in diagnostic and organizational practices. This study provides a descriptive analysis of patient trajectories for four major respiratory conditions: lung cancer, interstitial fibrosis, chronic obstructive pulmonary disease (COPD), and pneumonia. Trajectories were compared between a pre-COVID-19 period (2018&amp;amp;ndash;2019) and a COVID-19 period (2020&amp;amp;ndash;2022) in a specialized hospital. Using process mining applied to administrative event logs, we examined three aspects of care: the structure and sequencing of activities, the timing of transitions between care encounters, and imaging timeliness. The analysis spanned inpatient, emergency department, and outpatient settings. Indicators of care duration and transition timing revealed heterogeneous temporal patterns. Several conditions showed shorter intervals in the COVID-19 period, whereas others varied little. Activity-level analyses complemented these findings. Process maps indicated stable structural components in many pathways, together with differences in timing and execution. In the emergency department, care shifted toward bedside radiography, whereas CT chest volumes remained relatively stable across periods and settings. Imaging timeliness stayed consistently high in the emergency department and relatively stable for most inpatient conditions. Outcome-related indicators, including 30-day readmission and prolonged care trajectories, showed only modest differences between periods. Overall, the study demonstrates the value of process mining for describing real-world clinical pathways and identifying temporal variations in care. These results provide a foundation for future work that integrates richer clinical information and analytical approaches capable of assessing causal relationships.</description>
	<pubDate>2026-08-10</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 266: Descriptive Process Mining of Pulmonary Clinical Pathways Before and During COVID-19</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/8/266">doi: 10.3390/bdcc10080266</a></p>
	<p>Authors:
		Luca Murazzano
		Paolo Landa
		Jean-Baptiste Gartner
		André Côté
		</p>
	<p>Understanding how clinical pathways evolve over time is essential for characterizing care processes. It also helps identify potential shifts in diagnostic and organizational practices. This study provides a descriptive analysis of patient trajectories for four major respiratory conditions: lung cancer, interstitial fibrosis, chronic obstructive pulmonary disease (COPD), and pneumonia. Trajectories were compared between a pre-COVID-19 period (2018&amp;amp;ndash;2019) and a COVID-19 period (2020&amp;amp;ndash;2022) in a specialized hospital. Using process mining applied to administrative event logs, we examined three aspects of care: the structure and sequencing of activities, the timing of transitions between care encounters, and imaging timeliness. The analysis spanned inpatient, emergency department, and outpatient settings. Indicators of care duration and transition timing revealed heterogeneous temporal patterns. Several conditions showed shorter intervals in the COVID-19 period, whereas others varied little. Activity-level analyses complemented these findings. Process maps indicated stable structural components in many pathways, together with differences in timing and execution. In the emergency department, care shifted toward bedside radiography, whereas CT chest volumes remained relatively stable across periods and settings. Imaging timeliness stayed consistently high in the emergency department and relatively stable for most inpatient conditions. Outcome-related indicators, including 30-day readmission and prolonged care trajectories, showed only modest differences between periods. Overall, the study demonstrates the value of process mining for describing real-world clinical pathways and identifying temporal variations in care. These results provide a foundation for future work that integrates richer clinical information and analytical approaches capable of assessing causal relationships.</p>
	]]></content:encoded>

	<dc:title>Descriptive Process Mining of Pulmonary Clinical Pathways Before and During COVID-19</dc:title>
			<dc:creator>Luca Murazzano</dc:creator>
			<dc:creator>Paolo Landa</dc:creator>
			<dc:creator>Jean-Baptiste Gartner</dc:creator>
			<dc:creator>André Côté</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10080266</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-08-10</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-08-10</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>8</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>266</prism:startingPage>
		<prism:doi>10.3390/bdcc10080266</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/8/266</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/8/265">

	<title>BDCC, Vol. 10, Pages 265: Joint MLP and Token Pruning for Personalizing Vision Transformers</title>
	<link>https://www.mdpi.com/2504-2289/10/8/265</link>
	<description>ViTs have achieved excellent performance in image recognition tasks, but their large parameter counts and high computational complexity limit their deployment on resource-constrained devices. Most existing ViT pruning methods adopt class-agnostic pruning strategies, which fail to distinguish the diverse structural requirements of different target classes. As a result, they are prone to removing critical features, leading to class-wise accuracy imbalance in practical deployment. To address this issue, this paper proposes a class-aware joint pruning framework for ViTs, which collaboratively compresses the model from two orthogonal dimensions: MLP neurons and visual tokens. Specifically, (1) based on first-order Taylor expansion, we quantify the contribution of each MLP neuron to the target classes and adaptively prune redundant neurons to achieve structured compression, followed by lightweight fine-tuning on the target class subset; (2) we propose a Class-Guided Token Selection (CGTS) method, which constructs class prototype vectors using a few support samples of the target classes and then dynamically selects patch tokens that are semantically highly relevant to the target classes during inference in a zero-shot manner, requiring no additional training or fine-tuning. The two modules complement each other, achieving dual compression from the parameter dimension and the inference data dimension. Experiments on CIFAR-100 and TinyImageNet datasets using DeiT-Tiny/Small models demonstrate that, compared with state-of-the-art pruning methods, our method reduces GMACs on target class subsets by up to 48%, improves inference speed by nearly 50%, and requires only 0.8 KB of additional storage overhead per subset, ultimately achieving a superior trade-off among accuracy, computational efficiency, and storage overhead.</description>
	<pubDate>2026-08-09</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 265: Joint MLP and Token Pruning for Personalizing Vision Transformers</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/8/265">doi: 10.3390/bdcc10080265</a></p>
	<p>Authors:
		Zhiyue Li
		Tong Liu
		Feng Huang
		Xinzhi Huang
		Zhihao Zou
		</p>
	<p>ViTs have achieved excellent performance in image recognition tasks, but their large parameter counts and high computational complexity limit their deployment on resource-constrained devices. Most existing ViT pruning methods adopt class-agnostic pruning strategies, which fail to distinguish the diverse structural requirements of different target classes. As a result, they are prone to removing critical features, leading to class-wise accuracy imbalance in practical deployment. To address this issue, this paper proposes a class-aware joint pruning framework for ViTs, which collaboratively compresses the model from two orthogonal dimensions: MLP neurons and visual tokens. Specifically, (1) based on first-order Taylor expansion, we quantify the contribution of each MLP neuron to the target classes and adaptively prune redundant neurons to achieve structured compression, followed by lightweight fine-tuning on the target class subset; (2) we propose a Class-Guided Token Selection (CGTS) method, which constructs class prototype vectors using a few support samples of the target classes and then dynamically selects patch tokens that are semantically highly relevant to the target classes during inference in a zero-shot manner, requiring no additional training or fine-tuning. The two modules complement each other, achieving dual compression from the parameter dimension and the inference data dimension. Experiments on CIFAR-100 and TinyImageNet datasets using DeiT-Tiny/Small models demonstrate that, compared with state-of-the-art pruning methods, our method reduces GMACs on target class subsets by up to 48%, improves inference speed by nearly 50%, and requires only 0.8 KB of additional storage overhead per subset, ultimately achieving a superior trade-off among accuracy, computational efficiency, and storage overhead.</p>
	]]></content:encoded>

	<dc:title>Joint MLP and Token Pruning for Personalizing Vision Transformers</dc:title>
			<dc:creator>Zhiyue Li</dc:creator>
			<dc:creator>Tong Liu</dc:creator>
			<dc:creator>Feng Huang</dc:creator>
			<dc:creator>Xinzhi Huang</dc:creator>
			<dc:creator>Zhihao Zou</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10080265</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-08-09</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-08-09</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>8</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>265</prism:startingPage>
		<prism:doi>10.3390/bdcc10080265</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/8/265</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/8/264">

	<title>BDCC, Vol. 10, Pages 264: A Three-Stage Cross-Lingual Knowledge Transfer Approach Based on the XLM-RoBERTa Model for Detecting Fake News in Ukrainian</title>
	<link>https://www.mdpi.com/2504-2289/10/8/264</link>
	<description>In recent years, there has been an increase in the amount of fake news in the media, which is why fact-checking systems are gaining popularity, particularly those that use natural language processing (NLP) to quickly identify and flag fake news. One of the main limitations in the development of such systems is the limited number of datasets containing verified information, which are necessary for the effective training of models. The situation is particularly critical for non-English datasets, specifically those in the Ukrainian language. This article proposes a three-stage algorithm for training a model to recognize fake news in the Ukrainian language. At the core of the proposed approach lies the multilingual transformer model XLM-RoBERTa, which solves this problem by utilizing cross-lingual knowledge transfer from English to Ukrainian. This approach means there is no need to search for a large, high-quality dataset in Ukrainian; instead, a significantly smaller dataset in Ukrainian can be used for the final calibration of the model. The model developed as a result of the experiment proved effective in extreme low-resource scenarios, achieving 90.7% accuracy on just 500 training records and outperforming the baseline model by 9.7%.</description>
	<pubDate>2026-08-08</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 264: A Three-Stage Cross-Lingual Knowledge Transfer Approach Based on the XLM-RoBERTa Model for Detecting Fake News in Ukrainian</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/8/264">doi: 10.3390/bdcc10080264</a></p>
	<p>Authors:
		Volodymyr Smahliuk
		Yaroslav Kovivchak
		Yurii Kynash
		</p>
	<p>In recent years, there has been an increase in the amount of fake news in the media, which is why fact-checking systems are gaining popularity, particularly those that use natural language processing (NLP) to quickly identify and flag fake news. One of the main limitations in the development of such systems is the limited number of datasets containing verified information, which are necessary for the effective training of models. The situation is particularly critical for non-English datasets, specifically those in the Ukrainian language. This article proposes a three-stage algorithm for training a model to recognize fake news in the Ukrainian language. At the core of the proposed approach lies the multilingual transformer model XLM-RoBERTa, which solves this problem by utilizing cross-lingual knowledge transfer from English to Ukrainian. This approach means there is no need to search for a large, high-quality dataset in Ukrainian; instead, a significantly smaller dataset in Ukrainian can be used for the final calibration of the model. The model developed as a result of the experiment proved effective in extreme low-resource scenarios, achieving 90.7% accuracy on just 500 training records and outperforming the baseline model by 9.7%.</p>
	]]></content:encoded>

	<dc:title>A Three-Stage Cross-Lingual Knowledge Transfer Approach Based on the XLM-RoBERTa Model for Detecting Fake News in Ukrainian</dc:title>
			<dc:creator>Volodymyr Smahliuk</dc:creator>
			<dc:creator>Yaroslav Kovivchak</dc:creator>
			<dc:creator>Yurii Kynash</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10080264</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-08-08</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-08-08</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>8</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>264</prism:startingPage>
		<prism:doi>10.3390/bdcc10080264</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/8/264</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/8/263">

	<title>BDCC, Vol. 10, Pages 263: GluKDnet: A Lightweight Blood Glucose Prediction Model Based on Heterogeneous Knowledge Distillation</title>
	<link>https://www.mdpi.com/2504-2289/10/8/263</link>
	<description>Accurate blood glucose prediction is essential for glycemic management in people with diabetes, but the size of many high-performing models complicates execution on resource-constrained artificial pancreas controllers. We propose GluKDnet, a lightweight glucose-forecasting model for prospective Android-smartphone-based mobile edge controllers. GluKDnet transfers the representational capacity of a time-series foundation model to a compact causal CNN through heterogeneous knowledge distillation. The teacher model, MOMENT, is adapted to continuous glucose monitoring (CGM) data through risk-event-aware masking, which prioritizes abnormal glucose levels, rapid glucose fluctuations, and CGM-defined dawn phenomenon and Somogyi effect patterns during masked reconstruction. A transient-state and steady-state distillation module jointly aligns ordered patch-level dynamics and day-level summaries between teacher and student. Using DLCP3 for teacher pretraining and leave-one-patient-out evaluation on OhioT1DM, GluKDnet achieves RMSE values of 20.04, 32.04, and 45.33 mg/dL for 30, 60, and 120 min prediction, respectively, with about 53K parameters. Auxiliary evaluation on T1D-UoM shows a similar offline accuracy&amp;amp;ndash;parameter count pattern. On a vivo V2072A Android smartphone, the 30 min model achieved a mean API inference latency of 0.470 ms (P95: 0.855 ms), a maximum sampled process proportional-set-size memory of 47.06 MiB, and a median incremental device energy estimate of 0.277 mJ per inference. These device measurements characterize the exported student model under one hardware and software configuration; insulin dosing and prospective closed-loop clinical evaluation remain outside the scope of this study.</description>
	<pubDate>2026-08-06</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 263: GluKDnet: A Lightweight Blood Glucose Prediction Model Based on Heterogeneous Knowledge Distillation</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/8/263">doi: 10.3390/bdcc10080263</a></p>
	<p>Authors:
		Aowei Teng
		Xiaoyu Sun
		Hongru Li
		Xia Yu
		</p>
	<p>Accurate blood glucose prediction is essential for glycemic management in people with diabetes, but the size of many high-performing models complicates execution on resource-constrained artificial pancreas controllers. We propose GluKDnet, a lightweight glucose-forecasting model for prospective Android-smartphone-based mobile edge controllers. GluKDnet transfers the representational capacity of a time-series foundation model to a compact causal CNN through heterogeneous knowledge distillation. The teacher model, MOMENT, is adapted to continuous glucose monitoring (CGM) data through risk-event-aware masking, which prioritizes abnormal glucose levels, rapid glucose fluctuations, and CGM-defined dawn phenomenon and Somogyi effect patterns during masked reconstruction. A transient-state and steady-state distillation module jointly aligns ordered patch-level dynamics and day-level summaries between teacher and student. Using DLCP3 for teacher pretraining and leave-one-patient-out evaluation on OhioT1DM, GluKDnet achieves RMSE values of 20.04, 32.04, and 45.33 mg/dL for 30, 60, and 120 min prediction, respectively, with about 53K parameters. Auxiliary evaluation on T1D-UoM shows a similar offline accuracy&amp;amp;ndash;parameter count pattern. On a vivo V2072A Android smartphone, the 30 min model achieved a mean API inference latency of 0.470 ms (P95: 0.855 ms), a maximum sampled process proportional-set-size memory of 47.06 MiB, and a median incremental device energy estimate of 0.277 mJ per inference. These device measurements characterize the exported student model under one hardware and software configuration; insulin dosing and prospective closed-loop clinical evaluation remain outside the scope of this study.</p>
	]]></content:encoded>

	<dc:title>GluKDnet: A Lightweight Blood Glucose Prediction Model Based on Heterogeneous Knowledge Distillation</dc:title>
			<dc:creator>Aowei Teng</dc:creator>
			<dc:creator>Xiaoyu Sun</dc:creator>
			<dc:creator>Hongru Li</dc:creator>
			<dc:creator>Xia Yu</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10080263</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-08-06</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-08-06</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>8</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>263</prism:startingPage>
		<prism:doi>10.3390/bdcc10080263</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/8/263</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/8/262">

	<title>BDCC, Vol. 10, Pages 262: Secure Knowledge Retrieval for English-Teaching Agents: A Multi-Stage Auditing and Knowledge Purification Method</title>
	<link>https://www.mdpi.com/2504-2289/10/8/262</link>
	<description>English-teaching agents use external knowledge retrieval to update instructional content, broaden domain coverage, and personalize support beyond standalone large language models (LLMs). However, open sources may introduce harmful, biased, or misleading content into retrieval-augmented generation (RAG) pipelines, affecting learners&amp;amp;rsquo; judgment, cultural understanding, and value formation. To address this problem, this study proposes a multi-stage secure knowledge retrieval method for English-teaching agents. The method coordinates safeguards across knowledge-source access, retrieval execution, and model output. At the access stage, custom rules and Semgrep-based static scanning perform preliminary risk screening. At the retrieval stage, LLM-based dynamic evaluation identifies tool-description contamination and cross-file data-flow risks. At the output stage, semantic-embedding pre-screening, LLM review, and bounded knowledge purification detect and rewrite risky responses. Our experiments use public safety benchmarks, a mixed corpus of benign and poisoned passages, synthetic purification cases, and controlled end-to-end teaching scenarios. Compared with vanilla RAG, the framework reduces Poison Exposure@5 from 92.0% to 3.0% and retrieval attack success from 86.0% to 2.0% while preserving retrieval coverage. These results provide preliminary evidence that the framework can empower English teaching by enabling agents to deliver safer materials and trustworthy support for classroom questioning, academic writing, and intercultural learning.</description>
	<pubDate>2026-08-06</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 262: Secure Knowledge Retrieval for English-Teaching Agents: A Multi-Stage Auditing and Knowledge Purification Method</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/8/262">doi: 10.3390/bdcc10080262</a></p>
	<p>Authors:
		Jiming Yin
		Xianfeng Xie
		Shanyi Guo
		Jiawei Chen
		Jie Cui
		</p>
	<p>English-teaching agents use external knowledge retrieval to update instructional content, broaden domain coverage, and personalize support beyond standalone large language models (LLMs). However, open sources may introduce harmful, biased, or misleading content into retrieval-augmented generation (RAG) pipelines, affecting learners&amp;amp;rsquo; judgment, cultural understanding, and value formation. To address this problem, this study proposes a multi-stage secure knowledge retrieval method for English-teaching agents. The method coordinates safeguards across knowledge-source access, retrieval execution, and model output. At the access stage, custom rules and Semgrep-based static scanning perform preliminary risk screening. At the retrieval stage, LLM-based dynamic evaluation identifies tool-description contamination and cross-file data-flow risks. At the output stage, semantic-embedding pre-screening, LLM review, and bounded knowledge purification detect and rewrite risky responses. Our experiments use public safety benchmarks, a mixed corpus of benign and poisoned passages, synthetic purification cases, and controlled end-to-end teaching scenarios. Compared with vanilla RAG, the framework reduces Poison Exposure@5 from 92.0% to 3.0% and retrieval attack success from 86.0% to 2.0% while preserving retrieval coverage. These results provide preliminary evidence that the framework can empower English teaching by enabling agents to deliver safer materials and trustworthy support for classroom questioning, academic writing, and intercultural learning.</p>
	]]></content:encoded>

	<dc:title>Secure Knowledge Retrieval for English-Teaching Agents: A Multi-Stage Auditing and Knowledge Purification Method</dc:title>
			<dc:creator>Jiming Yin</dc:creator>
			<dc:creator>Xianfeng Xie</dc:creator>
			<dc:creator>Shanyi Guo</dc:creator>
			<dc:creator>Jiawei Chen</dc:creator>
			<dc:creator>Jie Cui</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10080262</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-08-06</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-08-06</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>8</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>262</prism:startingPage>
		<prism:doi>10.3390/bdcc10080262</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/8/262</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/8/261">

	<title>BDCC, Vol. 10, Pages 261: Cognitive Entanglement: Toward a Developmental Framework of the Human-AI Coevolutionary Leap</title>
	<link>https://www.mdpi.com/2504-2289/10/8/261</link>
	<description>Large language models have become routine participants in everyday cognition. Their role has widened from retrieval and text generation to helping users define problems, organize arguments, make judgments, and interpret themselves. Yet their cognitive consequences are strikingly divergent. For some users, generative AI appears to reduce critical engagement, independent judgment, and tolerance for difficulty. For others, the same class of systems becomes a medium for conceptual expansion, reflective questioning, and higher-order learning. This divergence cannot be explained by model capability alone. Mental effort is often treated as a cost to be reduced. Yet repeated delegation may also reduce opportunities to practice the processes required for independent judgment. The key issue is developmental: how sustained AI use changes users&amp;amp;rsquo; cognitive capacities over time. This perspective proposes cognitive entanglement as a framework for understanding the developmental consequences of sustained human-AI coupling. Cognitive entanglement refers to a relation in which human and AI activity become mutually shaping, irreducible to either party alone and organized across different developmental levels. The framework examines how repeated interaction with AI changes the ways users formulate problems, evaluate reasons, and make judgments. Unlike theories that locate the boundaries of cognition (the extended mind, enactivism) or explain the mechanisms of consciousness (global workspace, higher-order, predictive-processing, and integrated-information theories), cognitive entanglement examines whether sustained AI use preserves, weakens, or reorganizes users&amp;amp;rsquo; cognitive capacities. The article argues that current AI systems are often optimized for fluency, immediacy, and user satisfaction, and this may reduce the productive difficulty that supports higher-order cognitive development. If AI is to support human cognitive growth, design must move beyond answer provision and efficiency maximization toward the organization of productive human-AI relations: relations that challenge users&amp;amp;rsquo; initial assumptions while providing support appropriate to the task and the user&amp;amp;rsquo;s level of expertise. The argument draws on philosophy of mind, cognitive science, and learning science, and compares divergent approaches to coupling in order to specify which forms of relation carry which developmental consequences. The concept shifts attention from AI as a tool or automation system to the developmental consequences of sustained human-AI interaction.</description>
	<pubDate>2026-08-05</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 261: Cognitive Entanglement: Toward a Developmental Framework of the Human-AI Coevolutionary Leap</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/8/261">doi: 10.3390/bdcc10080261</a></p>
	<p>Authors:
		Xiao-Kun Wu
		Min Chen
		Giancarlo Fortino
		</p>
	<p>Large language models have become routine participants in everyday cognition. Their role has widened from retrieval and text generation to helping users define problems, organize arguments, make judgments, and interpret themselves. Yet their cognitive consequences are strikingly divergent. For some users, generative AI appears to reduce critical engagement, independent judgment, and tolerance for difficulty. For others, the same class of systems becomes a medium for conceptual expansion, reflective questioning, and higher-order learning. This divergence cannot be explained by model capability alone. Mental effort is often treated as a cost to be reduced. Yet repeated delegation may also reduce opportunities to practice the processes required for independent judgment. The key issue is developmental: how sustained AI use changes users&amp;amp;rsquo; cognitive capacities over time. This perspective proposes cognitive entanglement as a framework for understanding the developmental consequences of sustained human-AI coupling. Cognitive entanglement refers to a relation in which human and AI activity become mutually shaping, irreducible to either party alone and organized across different developmental levels. The framework examines how repeated interaction with AI changes the ways users formulate problems, evaluate reasons, and make judgments. Unlike theories that locate the boundaries of cognition (the extended mind, enactivism) or explain the mechanisms of consciousness (global workspace, higher-order, predictive-processing, and integrated-information theories), cognitive entanglement examines whether sustained AI use preserves, weakens, or reorganizes users&amp;amp;rsquo; cognitive capacities. The article argues that current AI systems are often optimized for fluency, immediacy, and user satisfaction, and this may reduce the productive difficulty that supports higher-order cognitive development. If AI is to support human cognitive growth, design must move beyond answer provision and efficiency maximization toward the organization of productive human-AI relations: relations that challenge users&amp;amp;rsquo; initial assumptions while providing support appropriate to the task and the user&amp;amp;rsquo;s level of expertise. The argument draws on philosophy of mind, cognitive science, and learning science, and compares divergent approaches to coupling in order to specify which forms of relation carry which developmental consequences. The concept shifts attention from AI as a tool or automation system to the developmental consequences of sustained human-AI interaction.</p>
	]]></content:encoded>

	<dc:title>Cognitive Entanglement: Toward a Developmental Framework of the Human-AI Coevolutionary Leap</dc:title>
			<dc:creator>Xiao-Kun Wu</dc:creator>
			<dc:creator>Min Chen</dc:creator>
			<dc:creator>Giancarlo Fortino</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10080261</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-08-05</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-08-05</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>8</prism:number>
	<prism:section>Perspective</prism:section>
	<prism:startingPage>261</prism:startingPage>
		<prism:doi>10.3390/bdcc10080261</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/8/261</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/8/260">

	<title>BDCC, Vol. 10, Pages 260: SETTA: Parameter-Free Test-Time Adaptation for Graph Neural Networks via Spectral-Energy-Guided Semantic Refinement</title>
	<link>https://www.mdpi.com/2504-2289/10/8/260</link>
	<description>Node classification is a central graph data mining task, yet repeated message passing can over-smooth representations and degrade frozen graph neural network (GNN) predictions after deployment. We present SETTA (Spectral-Energy Test-Time Adaptation), a prediction-level graph test-time adaptation framework that refines frozen outputs without test labels, gradients, parameter updates, or learnable adaptation parameters. SETTA denoises features for semantic-neighbor construction, adds complementary semantic routes while preserving observed edges, monitors a smoothness-energy proxy during diffusion, and accepts refinements through entropy-based gating. Configurations are fixed by a dataset-level protocol or selected using validation data only. Across six mostly homophilic benchmarks with 2708&amp;amp;ndash;19,717 nodes, SETTA improved a frozen two-layer GCN on every dataset and achieved the highest mean accuracy among the evaluated methods on five, with gains of 4.61, 3.08, and 2.01 percentage points on Cora, CiteSeer, and PubMed, respectively. Positive mean gains were also observed across all 30 dataset&amp;amp;ndash;backbone settings. Ablations and transition analyses indicate that semantic injection is most beneficial on sparse citation graphs and that selective refinement limits harmful changes. The current dense implementation supports benchmark-scale, amortized refinement; scalability and robustness on heterophilic graphs remain open.</description>
	<pubDate>2026-08-04</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 260: SETTA: Parameter-Free Test-Time Adaptation for Graph Neural Networks via Spectral-Energy-Guided Semantic Refinement</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/8/260">doi: 10.3390/bdcc10080260</a></p>
	<p>Authors:
		Dongyang Yu
		Xia Cui
		Rong Xiao
		</p>
	<p>Node classification is a central graph data mining task, yet repeated message passing can over-smooth representations and degrade frozen graph neural network (GNN) predictions after deployment. We present SETTA (Spectral-Energy Test-Time Adaptation), a prediction-level graph test-time adaptation framework that refines frozen outputs without test labels, gradients, parameter updates, or learnable adaptation parameters. SETTA denoises features for semantic-neighbor construction, adds complementary semantic routes while preserving observed edges, monitors a smoothness-energy proxy during diffusion, and accepts refinements through entropy-based gating. Configurations are fixed by a dataset-level protocol or selected using validation data only. Across six mostly homophilic benchmarks with 2708&amp;amp;ndash;19,717 nodes, SETTA improved a frozen two-layer GCN on every dataset and achieved the highest mean accuracy among the evaluated methods on five, with gains of 4.61, 3.08, and 2.01 percentage points on Cora, CiteSeer, and PubMed, respectively. Positive mean gains were also observed across all 30 dataset&amp;amp;ndash;backbone settings. Ablations and transition analyses indicate that semantic injection is most beneficial on sparse citation graphs and that selective refinement limits harmful changes. The current dense implementation supports benchmark-scale, amortized refinement; scalability and robustness on heterophilic graphs remain open.</p>
	]]></content:encoded>

	<dc:title>SETTA: Parameter-Free Test-Time Adaptation for Graph Neural Networks via Spectral-Energy-Guided Semantic Refinement</dc:title>
			<dc:creator>Dongyang Yu</dc:creator>
			<dc:creator>Xia Cui</dc:creator>
			<dc:creator>Rong Xiao</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10080260</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-08-04</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-08-04</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>8</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>260</prism:startingPage>
		<prism:doi>10.3390/bdcc10080260</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/8/260</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/8/259">

	<title>BDCC, Vol. 10, Pages 259: Adapting Large-Scale Foundation Models for Turkic Speech-to-Speech Translation: Fine-Tuned Cascade and Direct Approaches</title>
	<link>https://www.mdpi.com/2504-2289/10/8/259</link>
	<description>This paper presents a novel approach to speech-to-speech (STS) translation for low-resource Turkic languages. Today, STS has progressed rapidly for high-resource languages; the Turkic family remains significantly underrepresented. Consequently, it is quite challenging to develop a reliable speech translation system for these languages. To address this issue, we have developed two speech translation systems (STS) specifically tailored to Turkic languages with limited resources. The first, TurkicCascadeSTS, is a cascaded system that combines a fine-tuned Whisper-medium speech recognition model, GPT translation, and separate speech synthesis models for each language. The second system is a direct speech translation model based on a fine-tuned SeamlessM4Tv2. Both systems have been tested for translation into Turkic languages, using 24,656 audio recordings per language. The TurkicCascadeSTS system delivered far better results: the average BLEU score rose from 4.68 to 30.60; the METEOR score rose from 15.03 to 44.42; and the word error rate (WER) also fell significantly. These improvements are due to the fact that each module of the system was individually fine-tuned to account for the specific characteristics of each language. Although SeamlessM4Tv2 sometimes produces clearer audio, TurkicCascadeSTS generally delivers higher speech and translation quality for all language pairs. This demonstrates that modular, specially tuned systems are an effective solution for translation into Turkic languages, particularly given their complex structure and limited linguistic resources. Such systems could benefit more than 200 million native speakers of Turkic languages.</description>
	<pubDate>2026-08-04</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 259: Adapting Large-Scale Foundation Models for Turkic Speech-to-Speech Translation: Fine-Tuned Cascade and Direct Approaches</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/8/259">doi: 10.3390/bdcc10080259</a></p>
	<p>Authors:
		Aidana Karibayeva
		Vladislav Karyukin
		Oleg Myssov
		Dina Amirova
		Balzhan Abduali
		Adina Karybayeva
		</p>
	<p>This paper presents a novel approach to speech-to-speech (STS) translation for low-resource Turkic languages. Today, STS has progressed rapidly for high-resource languages; the Turkic family remains significantly underrepresented. Consequently, it is quite challenging to develop a reliable speech translation system for these languages. To address this issue, we have developed two speech translation systems (STS) specifically tailored to Turkic languages with limited resources. The first, TurkicCascadeSTS, is a cascaded system that combines a fine-tuned Whisper-medium speech recognition model, GPT translation, and separate speech synthesis models for each language. The second system is a direct speech translation model based on a fine-tuned SeamlessM4Tv2. Both systems have been tested for translation into Turkic languages, using 24,656 audio recordings per language. The TurkicCascadeSTS system delivered far better results: the average BLEU score rose from 4.68 to 30.60; the METEOR score rose from 15.03 to 44.42; and the word error rate (WER) also fell significantly. These improvements are due to the fact that each module of the system was individually fine-tuned to account for the specific characteristics of each language. Although SeamlessM4Tv2 sometimes produces clearer audio, TurkicCascadeSTS generally delivers higher speech and translation quality for all language pairs. This demonstrates that modular, specially tuned systems are an effective solution for translation into Turkic languages, particularly given their complex structure and limited linguistic resources. Such systems could benefit more than 200 million native speakers of Turkic languages.</p>
	]]></content:encoded>

	<dc:title>Adapting Large-Scale Foundation Models for Turkic Speech-to-Speech Translation: Fine-Tuned Cascade and Direct Approaches</dc:title>
			<dc:creator>Aidana Karibayeva</dc:creator>
			<dc:creator>Vladislav Karyukin</dc:creator>
			<dc:creator>Oleg Myssov</dc:creator>
			<dc:creator>Dina Amirova</dc:creator>
			<dc:creator>Balzhan Abduali</dc:creator>
			<dc:creator>Adina Karybayeva</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10080259</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-08-04</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-08-04</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>8</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>259</prism:startingPage>
		<prism:doi>10.3390/bdcc10080259</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/8/259</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/8/258">

	<title>BDCC, Vol. 10, Pages 258: A Governed NL-to-SQL Architecture for Reliable Clinical Data Querying and Outpatient Schedule Monitoring</title>
	<link>https://www.mdpi.com/2504-2289/10/8/258</link>
	<description>The deployment of natural-language-to-SQL (NL-to-SQL) systems in primary healthcare requires more than accurate query generation: it also requires governed data access, robustness to local terminology, and reliable handling of ambiguous user requests. This study evaluated a pilot proof-of-concept integrating a Spanish-language NL-to-SQL assistant with a governed, read-only outpatient scheduling repository derived from the Rayen information system used in a Centro de Salud Familiar (CESFAM) setting in Renca, Chile. The data used by the prototype were accessed through an external company responsible for data management in this context. The prototype was implemented with MindsDB as an artificial intelligence (AI)-enabled database layer and operated on anonymized, delayed secondary scheduling data. Evaluation was conducted through a Slack interface using 252 audited interactions from 42 users, with six assigned interactions per user and up to three exchanges per interaction. SQL correctness reached 240/252 (95.2%), whereas both query correctness and answer correctness reached 144/252 (57.1%). These findings suggest that governed pilot deployment for outpatient schedule monitoring may be feasible under controlled institutional conditions, while indicating that the main remaining barriers are semantic rather than purely syntactic, specifically ambiguity handling, institution-specific operational language, and faithful answer verbalization. The study therefore contributes deployment-oriented pilot evidence and clarifies where operational Spanish NL-to-SQL remains fragile under real institutional constraints.</description>
	<pubDate>2026-08-03</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 258: A Governed NL-to-SQL Architecture for Reliable Clinical Data Querying and Outpatient Schedule Monitoring</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/8/258">doi: 10.3390/bdcc10080258</a></p>
	<p>Authors:
		Isaac Daroch
		Matías Rojas Cabrera
		Rodrigo Muñoz Andrade
		Alejandra Fernández
		Juan Pablo Vásconez
		</p>
	<p>The deployment of natural-language-to-SQL (NL-to-SQL) systems in primary healthcare requires more than accurate query generation: it also requires governed data access, robustness to local terminology, and reliable handling of ambiguous user requests. This study evaluated a pilot proof-of-concept integrating a Spanish-language NL-to-SQL assistant with a governed, read-only outpatient scheduling repository derived from the Rayen information system used in a Centro de Salud Familiar (CESFAM) setting in Renca, Chile. The data used by the prototype were accessed through an external company responsible for data management in this context. The prototype was implemented with MindsDB as an artificial intelligence (AI)-enabled database layer and operated on anonymized, delayed secondary scheduling data. Evaluation was conducted through a Slack interface using 252 audited interactions from 42 users, with six assigned interactions per user and up to three exchanges per interaction. SQL correctness reached 240/252 (95.2%), whereas both query correctness and answer correctness reached 144/252 (57.1%). These findings suggest that governed pilot deployment for outpatient schedule monitoring may be feasible under controlled institutional conditions, while indicating that the main remaining barriers are semantic rather than purely syntactic, specifically ambiguity handling, institution-specific operational language, and faithful answer verbalization. The study therefore contributes deployment-oriented pilot evidence and clarifies where operational Spanish NL-to-SQL remains fragile under real institutional constraints.</p>
	]]></content:encoded>

	<dc:title>A Governed NL-to-SQL Architecture for Reliable Clinical Data Querying and Outpatient Schedule Monitoring</dc:title>
			<dc:creator>Isaac Daroch</dc:creator>
			<dc:creator>Matías Rojas Cabrera</dc:creator>
			<dc:creator>Rodrigo Muñoz Andrade</dc:creator>
			<dc:creator>Alejandra Fernández</dc:creator>
			<dc:creator>Juan Pablo Vásconez</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10080258</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-08-03</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-08-03</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>8</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>258</prism:startingPage>
		<prism:doi>10.3390/bdcc10080258</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/8/258</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/8/257">

	<title>BDCC, Vol. 10, Pages 257: PPRL-Stack: A Novel Stacking Architecture for Efficient and Secure Record Linkage</title>
	<link>https://www.mdpi.com/2504-2289/10/8/257</link>
	<description>Record linkage is a fundamental step in ensuring the quality of data by detecting duplicate records within different databases. Nevertheless, dealing with big, imbalanced databases and ensuring data confidentiality is still difficult in terms of performance and precision. This paper introduces a new Privacy-Preserving Record Linkage (PPRL) method named PPRL-Stack, which uses the Bloom filter encoding technique to hide information and a Stack Ensemble structure for classification. The proposed model consists of Support Vector Machine (SVM) as a base learner and Logistic Regression (LR) as a meta-classifier in combination with the application of Sorted Neighborhood Method (SNM) technique to bring down the time complexity to O(N log N). Experiments conducted on the Freely Extensible Biomedical Record Linkage (FEBRL) and North Carolina Voter Registration (NCVR) databases prove that the proposed PPRL-Stack can obtain nearly perfect discrimination with an F1-score of 0.9921. Particularly, our proposed architecture is more than 340 times and 40 times faster than the latest Siamese Bidirectional Long Short-Term Memory (Bi-LSTM) architecture in training and validation stages, respectively.</description>
	<pubDate>2026-08-03</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 257: PPRL-Stack: A Novel Stacking Architecture for Efficient and Secure Record Linkage</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/8/257">doi: 10.3390/bdcc10080257</a></p>
	<p>Authors:
		Fatima Zahrae Saber
		Ali Choukri
		Mohammed Amnai
		Abderrahim Waga
		</p>
	<p>Record linkage is a fundamental step in ensuring the quality of data by detecting duplicate records within different databases. Nevertheless, dealing with big, imbalanced databases and ensuring data confidentiality is still difficult in terms of performance and precision. This paper introduces a new Privacy-Preserving Record Linkage (PPRL) method named PPRL-Stack, which uses the Bloom filter encoding technique to hide information and a Stack Ensemble structure for classification. The proposed model consists of Support Vector Machine (SVM) as a base learner and Logistic Regression (LR) as a meta-classifier in combination with the application of Sorted Neighborhood Method (SNM) technique to bring down the time complexity to O(N log N). Experiments conducted on the Freely Extensible Biomedical Record Linkage (FEBRL) and North Carolina Voter Registration (NCVR) databases prove that the proposed PPRL-Stack can obtain nearly perfect discrimination with an F1-score of 0.9921. Particularly, our proposed architecture is more than 340 times and 40 times faster than the latest Siamese Bidirectional Long Short-Term Memory (Bi-LSTM) architecture in training and validation stages, respectively.</p>
	]]></content:encoded>

	<dc:title>PPRL-Stack: A Novel Stacking Architecture for Efficient and Secure Record Linkage</dc:title>
			<dc:creator>Fatima Zahrae Saber</dc:creator>
			<dc:creator>Ali Choukri</dc:creator>
			<dc:creator>Mohammed Amnai</dc:creator>
			<dc:creator>Abderrahim Waga</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10080257</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-08-03</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-08-03</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>8</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>257</prism:startingPage>
		<prism:doi>10.3390/bdcc10080257</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/8/257</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/8/256">

	<title>BDCC, Vol. 10, Pages 256: Cognition Orientation Risk Evaluation&amp;mdash;A Personality Driven Integrated Model for Phishing Susceptibility</title>
	<link>https://www.mdpi.com/2504-2289/10/8/256</link>
	<description>Phishing attacks exploit human vulnerabilities through social engineering techniques; therefore, a multidimensional framework integrating personality and cognition is crucial for tailoring personalized defenses against phishing threats. To prevent phishing attacks, focusing on psychological mechanisms has become the primary approach to address the human-centric nature of these threats. In response, we propose an integrated framework that synthesizes dimensions from the Five-Factor Model (FFM) and the Myers&amp;amp;ndash;Briggs Type Indicator (MBTI), grounded in Dual Process Theory to explore the cognitive drivers underlying decision-making under threat. To operationalize this framework, we developed a decision tree classifier to quantify the predictive significance of various personality traits and their hierarchical interactions. The results indicate that the personality types associated with the highest phishing risk profiles are ENFP, ESFP, ESTP, and ENFJ. These hierarchical classification results are projected onto the C.O.R.E. Quadrant (Cognition Orientation Risk Evaluation Quadrant) which enables the systematic representation of risk patterns across all 16 personality types. By providing a structured visualization of personality-driven risk patterns, the C.O.R.E. Quadrant offers a practical foundation for developing personalized defense mechanisms and tailored cybersecurity training strategies, moving beyond one-size-fits-all security protocols.</description>
	<pubDate>2026-08-03</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 256: Cognition Orientation Risk Evaluation&amp;mdash;A Personality Driven Integrated Model for Phishing Susceptibility</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/8/256">doi: 10.3390/bdcc10080256</a></p>
	<p>Authors:
		Chih-Hong Kao
		Chia-Wei Tsai
		Yu-Ting Kao
		Chao-Lung Chou
		</p>
	<p>Phishing attacks exploit human vulnerabilities through social engineering techniques; therefore, a multidimensional framework integrating personality and cognition is crucial for tailoring personalized defenses against phishing threats. To prevent phishing attacks, focusing on psychological mechanisms has become the primary approach to address the human-centric nature of these threats. In response, we propose an integrated framework that synthesizes dimensions from the Five-Factor Model (FFM) and the Myers&amp;amp;ndash;Briggs Type Indicator (MBTI), grounded in Dual Process Theory to explore the cognitive drivers underlying decision-making under threat. To operationalize this framework, we developed a decision tree classifier to quantify the predictive significance of various personality traits and their hierarchical interactions. The results indicate that the personality types associated with the highest phishing risk profiles are ENFP, ESFP, ESTP, and ENFJ. These hierarchical classification results are projected onto the C.O.R.E. Quadrant (Cognition Orientation Risk Evaluation Quadrant) which enables the systematic representation of risk patterns across all 16 personality types. By providing a structured visualization of personality-driven risk patterns, the C.O.R.E. Quadrant offers a practical foundation for developing personalized defense mechanisms and tailored cybersecurity training strategies, moving beyond one-size-fits-all security protocols.</p>
	]]></content:encoded>

	<dc:title>Cognition Orientation Risk Evaluation&amp;amp;mdash;A Personality Driven Integrated Model for Phishing Susceptibility</dc:title>
			<dc:creator>Chih-Hong Kao</dc:creator>
			<dc:creator>Chia-Wei Tsai</dc:creator>
			<dc:creator>Yu-Ting Kao</dc:creator>
			<dc:creator>Chao-Lung Chou</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10080256</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-08-03</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-08-03</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>8</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>256</prism:startingPage>
		<prism:doi>10.3390/bdcc10080256</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/8/256</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/8/255">

	<title>BDCC, Vol. 10, Pages 255: A Cloud-Based Multidimensional Big Data Framework for Healthcare Analytics: Bridging OLAP and Cognitive Insights in Chronic Pain Management</title>
	<link>https://www.mdpi.com/2504-2289/10/8/255</link>
	<description>By considering the real-life research project Pain-RELife, which focuses attention on big data management and analytics tools over patients suffering from chronic pain, located in the Region Lombardy of North Italy, this paper provides a relevant journey from theory to practice about so-called Multidimensional Big Data Analytics tools over (real-life) big healthcare datasets, as dictated by the effective project goals. This constitutes an effective contribution to the state-of-the-art research. Our conceptual and theoretical results are corroborated by a comprehensive campaign of real-life experimental results focused on advanced tools such as Multidimensional Clustering and Multidimensional Regression.</description>
	<pubDate>2026-08-02</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 255: A Cloud-Based Multidimensional Big Data Framework for Healthcare Analytics: Bridging OLAP and Cognitive Insights in Chronic Pain Management</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/8/255">doi: 10.3390/bdcc10080255</a></p>
	<p>Authors:
		Alfredo Cuzzocrea
		Abderraouf Hafsaoui
		</p>
	<p>By considering the real-life research project Pain-RELife, which focuses attention on big data management and analytics tools over patients suffering from chronic pain, located in the Region Lombardy of North Italy, this paper provides a relevant journey from theory to practice about so-called Multidimensional Big Data Analytics tools over (real-life) big healthcare datasets, as dictated by the effective project goals. This constitutes an effective contribution to the state-of-the-art research. Our conceptual and theoretical results are corroborated by a comprehensive campaign of real-life experimental results focused on advanced tools such as Multidimensional Clustering and Multidimensional Regression.</p>
	]]></content:encoded>

	<dc:title>A Cloud-Based Multidimensional Big Data Framework for Healthcare Analytics: Bridging OLAP and Cognitive Insights in Chronic Pain Management</dc:title>
			<dc:creator>Alfredo Cuzzocrea</dc:creator>
			<dc:creator>Abderraouf Hafsaoui</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10080255</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-08-02</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-08-02</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>8</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>255</prism:startingPage>
		<prism:doi>10.3390/bdcc10080255</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/8/255</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/8/254">

	<title>BDCC, Vol. 10, Pages 254: ExPAM: Explainable Personality Assessment Method Using Heterogeneous Linguistic Features and Off-the-Shelf LLMs</title>
	<link>https://www.mdpi.com/2504-2289/10/8/254</link>
	<description>Many organizations increasingly adopt personalization techniques to enhance user satisfaction. However, current systems generally cannot automatically infer and interpret individual personality traits (PTs), although these traits are key drivers of user behavior. While Large Language Models (LLMs) are widely used, they remain poorly suited to reliable and explainable Personality Assessment (PA). To address this gap, we propose ExPAM, a novel Explainable Personality Assessment Method that combines hybrid feature fusion with in-context learning in off-the-shelf LLMs to predict Big Five PTs from text. ExPAM explicitly grounds its predictions in interpretable linguistic patterns without requiring LLM fine-tuning. Its hybrid fusion is designed to improve both predictive performance and interpretability in PA. Transformer-based embeddings encode local contextual information, whereas features extracted using the Linguistic Inquiry and Word Count (LIWC) dictionary provide complementary global and local linguistic indicators of PTs. These interpretable feature patterns are included in prompts that guide the LLM to produce both PT predictions and human-understandable explanations. ExPAM shows competitive performance compared with multi-task models on the ChaLearn First Impressions v2 (FIv2) corpus and single-task models on the PANDORA corpus that rely on a single feature set. On FIv2, it achieves a mean accuracy (mAC) of 0.891 and a Concordance Correlation Coefficient (CCC) of 0.333. On PANDORA, it achieves a mean Pearson Correlation Coefficient (PCC) of 0.240 and a CCC of 0.101. Prompting the LLM with hybrid global&amp;amp;ndash;local patterns further improves CCC by 9.9% on FIv2 and 15.8% on PANDORA, while changes in mAC and mean PCC remain marginal. Qualitative interpretability analysis reveals trait-specific linguistic patterns, highlighting the potential of ExPAM for psychological research, computational linguistics, and paralinguistic studies.</description>
	<pubDate>2026-08-01</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 254: ExPAM: Explainable Personality Assessment Method Using Heterogeneous Linguistic Features and Off-the-Shelf LLMs</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/8/254">doi: 10.3390/bdcc10080254</a></p>
	<p>Authors:
		Elena Ryumina
		Dmitry Ryumin
		Maxim Markitantov
		Alexey Karpov
		</p>
	<p>Many organizations increasingly adopt personalization techniques to enhance user satisfaction. However, current systems generally cannot automatically infer and interpret individual personality traits (PTs), although these traits are key drivers of user behavior. While Large Language Models (LLMs) are widely used, they remain poorly suited to reliable and explainable Personality Assessment (PA). To address this gap, we propose ExPAM, a novel Explainable Personality Assessment Method that combines hybrid feature fusion with in-context learning in off-the-shelf LLMs to predict Big Five PTs from text. ExPAM explicitly grounds its predictions in interpretable linguistic patterns without requiring LLM fine-tuning. Its hybrid fusion is designed to improve both predictive performance and interpretability in PA. Transformer-based embeddings encode local contextual information, whereas features extracted using the Linguistic Inquiry and Word Count (LIWC) dictionary provide complementary global and local linguistic indicators of PTs. These interpretable feature patterns are included in prompts that guide the LLM to produce both PT predictions and human-understandable explanations. ExPAM shows competitive performance compared with multi-task models on the ChaLearn First Impressions v2 (FIv2) corpus and single-task models on the PANDORA corpus that rely on a single feature set. On FIv2, it achieves a mean accuracy (mAC) of 0.891 and a Concordance Correlation Coefficient (CCC) of 0.333. On PANDORA, it achieves a mean Pearson Correlation Coefficient (PCC) of 0.240 and a CCC of 0.101. Prompting the LLM with hybrid global&amp;amp;ndash;local patterns further improves CCC by 9.9% on FIv2 and 15.8% on PANDORA, while changes in mAC and mean PCC remain marginal. Qualitative interpretability analysis reveals trait-specific linguistic patterns, highlighting the potential of ExPAM for psychological research, computational linguistics, and paralinguistic studies.</p>
	]]></content:encoded>

	<dc:title>ExPAM: Explainable Personality Assessment Method Using Heterogeneous Linguistic Features and Off-the-Shelf LLMs</dc:title>
			<dc:creator>Elena Ryumina</dc:creator>
			<dc:creator>Dmitry Ryumin</dc:creator>
			<dc:creator>Maxim Markitantov</dc:creator>
			<dc:creator>Alexey Karpov</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10080254</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-08-01</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-08-01</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>8</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>254</prism:startingPage>
		<prism:doi>10.3390/bdcc10080254</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/8/254</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/8/253">

	<title>BDCC, Vol. 10, Pages 253: STDPatch: A Three-Stream Framework for Long-Term Time Series Forecasting via SG-Filter-Based Decomposition and Patch Refactor</title>
	<link>https://www.mdpi.com/2504-2289/10/8/253</link>
	<description>Driven by non-stationary factors in real-world sensor-driven applications, time series streams from energy meters, traffic detectors, weather stations, and industrial monitors often exhibit complex patterns composed of long-term trends and multi-scale seasonal fluctuations. Accurately disentangling and modeling these heterogeneous components remains a fundamental challenge in long-term time series forecasting (LTSF). To address this issue, we propose STDPatch, a novel three-stream forecasting framework that combines structural decomposition with architecture specialization. First, we introduce an SG-Filter-Based seasonal&amp;amp;ndash;trend decomposition module that employs polynomial fitting to extract shape-preserving trends while reducing seasonal noise. Second, we design a trend decomposition module that further separates the trend component into ascending and descending segments to capture fine-grained evolutionary dynamics. Third, we propose a patch refactor module that adaptively aggregates adjacent patches according to structural similarity, thereby preserving temporal semantic continuity and reducing spurious correlations. Finally, we develop a three-stream architecture that leverages convolutional, linear, and Transformer branches to model seasonal patterns, smooth trends, and non-stationary sub-trends, respectively, with each branch built from efficient, channel-independent components. Extensive experiments on seven real-world sensor-derived benchmark datasets demonstrate that STDPatch consistently outperforms state-of-the-art methods for long-term time series forecasting.</description>
	<pubDate>2026-08-01</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 253: STDPatch: A Three-Stream Framework for Long-Term Time Series Forecasting via SG-Filter-Based Decomposition and Patch Refactor</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/8/253">doi: 10.3390/bdcc10080253</a></p>
	<p>Authors:
		Lanlan Li
		Di Liu
		Shengfa Miao
		Ahmed Zahir
		Yongkang Mu
		Hualong Deng
		Xin Jin
		Qian Jiang
		Puming Wang
		Hua Jiang
		Shaowen Yao
		</p>
	<p>Driven by non-stationary factors in real-world sensor-driven applications, time series streams from energy meters, traffic detectors, weather stations, and industrial monitors often exhibit complex patterns composed of long-term trends and multi-scale seasonal fluctuations. Accurately disentangling and modeling these heterogeneous components remains a fundamental challenge in long-term time series forecasting (LTSF). To address this issue, we propose STDPatch, a novel three-stream forecasting framework that combines structural decomposition with architecture specialization. First, we introduce an SG-Filter-Based seasonal&amp;amp;ndash;trend decomposition module that employs polynomial fitting to extract shape-preserving trends while reducing seasonal noise. Second, we design a trend decomposition module that further separates the trend component into ascending and descending segments to capture fine-grained evolutionary dynamics. Third, we propose a patch refactor module that adaptively aggregates adjacent patches according to structural similarity, thereby preserving temporal semantic continuity and reducing spurious correlations. Finally, we develop a three-stream architecture that leverages convolutional, linear, and Transformer branches to model seasonal patterns, smooth trends, and non-stationary sub-trends, respectively, with each branch built from efficient, channel-independent components. Extensive experiments on seven real-world sensor-derived benchmark datasets demonstrate that STDPatch consistently outperforms state-of-the-art methods for long-term time series forecasting.</p>
	]]></content:encoded>

	<dc:title>STDPatch: A Three-Stream Framework for Long-Term Time Series Forecasting via SG-Filter-Based Decomposition and Patch Refactor</dc:title>
			<dc:creator>Lanlan Li</dc:creator>
			<dc:creator>Di Liu</dc:creator>
			<dc:creator>Shengfa Miao</dc:creator>
			<dc:creator>Ahmed Zahir</dc:creator>
			<dc:creator>Yongkang Mu</dc:creator>
			<dc:creator>Hualong Deng</dc:creator>
			<dc:creator>Xin Jin</dc:creator>
			<dc:creator>Qian Jiang</dc:creator>
			<dc:creator>Puming Wang</dc:creator>
			<dc:creator>Hua Jiang</dc:creator>
			<dc:creator>Shaowen Yao</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10080253</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-08-01</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-08-01</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>8</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>253</prism:startingPage>
		<prism:doi>10.3390/bdcc10080253</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/8/253</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/8/252">

	<title>BDCC, Vol. 10, Pages 252: A Systematic Review of Smart Home IoT Security: Applications, Threat Taxonomy, Privacy Risks, and Emerging Defensive Solutions</title>
	<link>https://www.mdpi.com/2504-2289/10/8/252</link>
	<description>The rapid proliferation of Internet of Things (IoT) technologies has transformed the modern home into a complex cyber&amp;amp;ndash;physical ecosystem encompassing hundreds of millions of connected devices globally. Smart homes support automation, energy management, and healthcare monitoring, but they also introduce a broad and evolving range of security and privacy challenges. This review examines 233 sources published between 2018 and May 2025, selected through a PRISMA-informed process covering five major academic databases and relevant standards and technical reports. It discusses communication protocols, including Matter, develops a Threat-Layer-Defense synthesis matrix covering ten attack categories; examines the practical limitations of AI-based anomaly detection and blockchain-based trust management; and derives recommendations for manufacturers, platform providers, users, and regulators. Privacy challenges, regulatory frameworks, and user behavior are considered alongside technical threats. The findings suggest that scalable smart home security requires coordinated progress in protocol standardization, enforceable device update lifecycles, gateway-level anomaly detection, and privacy-preserving local analytics rather than reliance on a single technical solution.</description>
	<pubDate>2026-08-01</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 252: A Systematic Review of Smart Home IoT Security: Applications, Threat Taxonomy, Privacy Risks, and Emerging Defensive Solutions</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/8/252">doi: 10.3390/bdcc10080252</a></p>
	<p>Authors:
		Dalibor Radovanovic
		Nikola Savanovic
		Jelena Janackovic
		Petar Kresoja
		</p>
	<p>The rapid proliferation of Internet of Things (IoT) technologies has transformed the modern home into a complex cyber&amp;amp;ndash;physical ecosystem encompassing hundreds of millions of connected devices globally. Smart homes support automation, energy management, and healthcare monitoring, but they also introduce a broad and evolving range of security and privacy challenges. This review examines 233 sources published between 2018 and May 2025, selected through a PRISMA-informed process covering five major academic databases and relevant standards and technical reports. It discusses communication protocols, including Matter, develops a Threat-Layer-Defense synthesis matrix covering ten attack categories; examines the practical limitations of AI-based anomaly detection and blockchain-based trust management; and derives recommendations for manufacturers, platform providers, users, and regulators. Privacy challenges, regulatory frameworks, and user behavior are considered alongside technical threats. The findings suggest that scalable smart home security requires coordinated progress in protocol standardization, enforceable device update lifecycles, gateway-level anomaly detection, and privacy-preserving local analytics rather than reliance on a single technical solution.</p>
	]]></content:encoded>

	<dc:title>A Systematic Review of Smart Home IoT Security: Applications, Threat Taxonomy, Privacy Risks, and Emerging Defensive Solutions</dc:title>
			<dc:creator>Dalibor Radovanovic</dc:creator>
			<dc:creator>Nikola Savanovic</dc:creator>
			<dc:creator>Jelena Janackovic</dc:creator>
			<dc:creator>Petar Kresoja</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10080252</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-08-01</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-08-01</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>8</prism:number>
	<prism:section>Systematic Review</prism:section>
	<prism:startingPage>252</prism:startingPage>
		<prism:doi>10.3390/bdcc10080252</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/8/252</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/8/251">

	<title>BDCC, Vol. 10, Pages 251: Beyond Transcript Alignment: Diagnosing Paralinguistic Information Flow in Frozen Speech-to-LLM Adapters</title>
	<link>https://www.mdpi.com/2504-2289/10/8/251</link>
	<description>Frozen speech-to-LLM systems train an adapter between a frozen audio encoder and a text LLM. We test whether adapters preserve sentence stress beyond transcripts and whether the LLM uses it. In a five-seed Qwen3-8B/WavLM baseline, transcript alignment lowers linear adapter Probe-K below the text-only KT baseline (0.211 vs. 0.290). The R1.8 configuration improves MLP-2 Probe-K from 0.245 to 0.306 (paired p=0.012) and passes three controls, but because warmup and augmentation also change, the gain is not attributed to Lcf alone; its crossing of the linear KT floor is neither statistically established nor capacity-matched. Response Probe-G remains 0.512 in both cohorts; single-seed LoRA and styled-teacher pilots do not improve it. A post hoc trace through all 36 Qwen blocks finds persistent absolute stress decodability in speech-slot states, but no reliable R1.8-over-R0 advantage at any state and no speech-conditioned answer margin; an explicit text tag instead yields a +6.40-nat margin. Attention differences depend on slot-length normalization, and late left/system concentration is shared across modalities rather than speech-specific. Thus, the tested system remains an end-to-end negative: adapter decodability does not imply causal response use. Scope is limited to sentence stress, mostly synthetic voices, one encoder, and one LLM.</description>
	<pubDate>2026-07-30</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 251: Beyond Transcript Alignment: Diagnosing Paralinguistic Information Flow in Frozen Speech-to-LLM Adapters</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/8/251">doi: 10.3390/bdcc10080251</a></p>
	<p>Authors:
		Nurgali Kadyrbek
		Madina Mansurova
		</p>
	<p>Frozen speech-to-LLM systems train an adapter between a frozen audio encoder and a text LLM. We test whether adapters preserve sentence stress beyond transcripts and whether the LLM uses it. In a five-seed Qwen3-8B/WavLM baseline, transcript alignment lowers linear adapter Probe-K below the text-only KT baseline (0.211 vs. 0.290). The R1.8 configuration improves MLP-2 Probe-K from 0.245 to 0.306 (paired p=0.012) and passes three controls, but because warmup and augmentation also change, the gain is not attributed to Lcf alone; its crossing of the linear KT floor is neither statistically established nor capacity-matched. Response Probe-G remains 0.512 in both cohorts; single-seed LoRA and styled-teacher pilots do not improve it. A post hoc trace through all 36 Qwen blocks finds persistent absolute stress decodability in speech-slot states, but no reliable R1.8-over-R0 advantage at any state and no speech-conditioned answer margin; an explicit text tag instead yields a +6.40-nat margin. Attention differences depend on slot-length normalization, and late left/system concentration is shared across modalities rather than speech-specific. Thus, the tested system remains an end-to-end negative: adapter decodability does not imply causal response use. Scope is limited to sentence stress, mostly synthetic voices, one encoder, and one LLM.</p>
	]]></content:encoded>

	<dc:title>Beyond Transcript Alignment: Diagnosing Paralinguistic Information Flow in Frozen Speech-to-LLM Adapters</dc:title>
			<dc:creator>Nurgali Kadyrbek</dc:creator>
			<dc:creator>Madina Mansurova</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10080251</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-07-30</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-07-30</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>8</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>251</prism:startingPage>
		<prism:doi>10.3390/bdcc10080251</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/8/251</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/8/250">

	<title>BDCC, Vol. 10, Pages 250: A Novel Multimodal Hierarchical Large Language Model for Enhancing Image and Text Sequence Recommendations in Item and User Modeling</title>
	<link>https://www.mdpi.com/2504-2289/10/8/250</link>
	<description>Existing e-commerce recommender systems, like graph-based retrieval and user behavior-driven Collaborative Filtering, usually take either textual or visual features for item recommendations to users. Traditional hierarchical large language models (HLLM) also suffer from high computational overhead for long user sequences and insufficient cross-modal semantic alignment. In this paper, we propose a multimodal hierarchical large language model (MHLLM) recommender system that leverages user behavior and integrates large language models to enhance text&amp;amp;ndash;image search precision and commercial value. Our MHLLM is a two-stage V-shaped model. The model decouples multimodal feature modeling and user behavior modeling: the first stage leverages Item-LLM and Item-CLIP to extract text semantics and cross-modal visual-text features respectively; the second stage adopts a learnable dynamic gating mechanism to adaptively fuse dual-modal features and uses User-LLM to model user preferences for personalized recommendation. We further design a multi-phase training strategy and lightweight feature projection structure to solve the gradient vanishing problem in joint training and reduce computational cost. Experimental results show that our MHLLM significantly outperforms traditional systems, improving key metrics like Recall@5 and NDCG@5 over 20% in most cases. Additionally, by efficient leveraging LLM and CLIP, MHLLM reduces the repeated inference overhead of User-LLM by over 30% via feature caching. Ablation experiments verify the effectiveness of core components such as dynamic gating, gating smoothing loss and task-specific prompts. The proposed model balances recommendation accuracy and deployment efficiency and has practical application value for large-scale e-commerce image-text retrieval and recommendation scenarios.</description>
	<pubDate>2026-07-29</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 250: A Novel Multimodal Hierarchical Large Language Model for Enhancing Image and Text Sequence Recommendations in Item and User Modeling</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/8/250">doi: 10.3390/bdcc10080250</a></p>
	<p>Authors:
		Wen-Long Dong
		Richard Tai-Chiu Hsung
		Harris Sik-Ho Tsang
		Tony Yulin Zhu
		Wai-Lun Lo
		</p>
	<p>Existing e-commerce recommender systems, like graph-based retrieval and user behavior-driven Collaborative Filtering, usually take either textual or visual features for item recommendations to users. Traditional hierarchical large language models (HLLM) also suffer from high computational overhead for long user sequences and insufficient cross-modal semantic alignment. In this paper, we propose a multimodal hierarchical large language model (MHLLM) recommender system that leverages user behavior and integrates large language models to enhance text&amp;amp;ndash;image search precision and commercial value. Our MHLLM is a two-stage V-shaped model. The model decouples multimodal feature modeling and user behavior modeling: the first stage leverages Item-LLM and Item-CLIP to extract text semantics and cross-modal visual-text features respectively; the second stage adopts a learnable dynamic gating mechanism to adaptively fuse dual-modal features and uses User-LLM to model user preferences for personalized recommendation. We further design a multi-phase training strategy and lightweight feature projection structure to solve the gradient vanishing problem in joint training and reduce computational cost. Experimental results show that our MHLLM significantly outperforms traditional systems, improving key metrics like Recall@5 and NDCG@5 over 20% in most cases. Additionally, by efficient leveraging LLM and CLIP, MHLLM reduces the repeated inference overhead of User-LLM by over 30% via feature caching. Ablation experiments verify the effectiveness of core components such as dynamic gating, gating smoothing loss and task-specific prompts. The proposed model balances recommendation accuracy and deployment efficiency and has practical application value for large-scale e-commerce image-text retrieval and recommendation scenarios.</p>
	]]></content:encoded>

	<dc:title>A Novel Multimodal Hierarchical Large Language Model for Enhancing Image and Text Sequence Recommendations in Item and User Modeling</dc:title>
			<dc:creator>Wen-Long Dong</dc:creator>
			<dc:creator>Richard Tai-Chiu Hsung</dc:creator>
			<dc:creator>Harris Sik-Ho Tsang</dc:creator>
			<dc:creator>Tony Yulin Zhu</dc:creator>
			<dc:creator>Wai-Lun Lo</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10080250</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-07-29</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-07-29</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>8</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>250</prism:startingPage>
		<prism:doi>10.3390/bdcc10080250</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/8/250</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/8/249">

	<title>BDCC, Vol. 10, Pages 249: From &amp;ldquo;Moneyball&amp;rdquo; to &amp;ldquo;Sports Bra&amp;rdquo;: A Qualitative Interview Study on the Use of Cognitive Computing Systems in Sports</title>
	<link>https://www.mdpi.com/2504-2289/10/8/249</link>
	<description>There are a wide range of possible applications for the collection and analysis of statistical data in sports, although their potential has not yet been fully exploited. This study focuses on the areas in which cognitive computing systems can offer advantages for sports organizations. Furthermore, it explores the extent to which media companies can use artificial intelligence to evaluate unstructured data and provide better services. Six semi-structured interviews with experts from the fields of sports, media, and information technology were evaluated using qualitative content analysis. This revealed the need for companies in both industries to adapt to rapidly changing market conditions. The speed of decision-making can be increased by collecting and analyzing large amounts of data in real time. Furthermore, relationships can be derived that were previously hidden due to the cognitive limitations of the human brain. Based on the analysis of all dimensions of athletic ability and the facets of a player&amp;amp;rsquo;s character, team performance can be improved. In addition, it is possible to assess the extent to which an athlete&amp;amp;rsquo;s character is compatible with a team and with which teammates he or she is likely to be a better or worse fit. Media companies are enabled to provide sports organizations with insights from the use of cognitive applications and are transforming from media companies to service providers.</description>
	<pubDate>2026-07-28</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 249: From &amp;ldquo;Moneyball&amp;rdquo; to &amp;ldquo;Sports Bra&amp;rdquo;: A Qualitative Interview Study on the Use of Cognitive Computing Systems in Sports</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/8/249">doi: 10.3390/bdcc10080249</a></p>
	<p>Authors:
		Sören Bär
		Yannick Wagner
		Markus Kurscheidt
		</p>
	<p>There are a wide range of possible applications for the collection and analysis of statistical data in sports, although their potential has not yet been fully exploited. This study focuses on the areas in which cognitive computing systems can offer advantages for sports organizations. Furthermore, it explores the extent to which media companies can use artificial intelligence to evaluate unstructured data and provide better services. Six semi-structured interviews with experts from the fields of sports, media, and information technology were evaluated using qualitative content analysis. This revealed the need for companies in both industries to adapt to rapidly changing market conditions. The speed of decision-making can be increased by collecting and analyzing large amounts of data in real time. Furthermore, relationships can be derived that were previously hidden due to the cognitive limitations of the human brain. Based on the analysis of all dimensions of athletic ability and the facets of a player&amp;amp;rsquo;s character, team performance can be improved. In addition, it is possible to assess the extent to which an athlete&amp;amp;rsquo;s character is compatible with a team and with which teammates he or she is likely to be a better or worse fit. Media companies are enabled to provide sports organizations with insights from the use of cognitive applications and are transforming from media companies to service providers.</p>
	]]></content:encoded>

	<dc:title>From &amp;amp;ldquo;Moneyball&amp;amp;rdquo; to &amp;amp;ldquo;Sports Bra&amp;amp;rdquo;: A Qualitative Interview Study on the Use of Cognitive Computing Systems in Sports</dc:title>
			<dc:creator>Sören Bär</dc:creator>
			<dc:creator>Yannick Wagner</dc:creator>
			<dc:creator>Markus Kurscheidt</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10080249</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-07-28</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-07-28</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>8</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>249</prism:startingPage>
		<prism:doi>10.3390/bdcc10080249</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/8/249</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/8/248">

	<title>BDCC, Vol. 10, Pages 248: RACPR: Generative AI-Enhanced Risk-Aware Causal Path Re-Ranking for Interpretable Multimorbidity Risk Identification in Elderly Health Consultation Scenarios</title>
	<link>https://www.mdpi.com/2504-2289/10/8/248</link>
	<description>Multimorbidity risk identification in elderly health consultation scenarios is challenging because chronic disease history and medication exposure may interact through complex causal pathways. Existing LLM-, RAG-, and graph-based retrieval methods often rely on semantic relevance or graph connectivity, which may be insufficient for identifying user-specific risk mechanisms. This study proposes RACPR, a generative AI-enhanced risk-aware causal path re-ranking framework for interpretable multimorbidity risk identification. RACPR ranks candidate causal paths by integrating user-entity alignment, causal coherence, risk contribution, and path length control and uses generative AI to transform selected paths into readable, path-grounded explanations. To support controlled algorithmic evaluation, we constructed a normalized benchmark of 1002 elderly multimorbidity consultation cases covering diabetes, hypertension, and chronic kidney disease. The benchmark and supporting knowledge graph were derived from publicly available biomedical and health information resources, including PubMed abstracts, guideline and review sources, DrugBank medication-safety evidence, and MedlinePlus-based terminology. Under a leakage-controlled setting, only age, diagnosed diseases, and medication exposures were used as model-accessible inputs, while abnormal indicators, support paths, and rationales were reserved for evaluation. On the test set, RACPR achieved an Accuracy of 0.659, a Precision of 0.602, a Recall of 0.938, an F1-score of 0.733, and an AUC of 0.777. Ablation analysis showed that removing the risk-aware component reduced AUC from 0.777 to 0.428. Benchmark-level explanation evaluation further showed improvements in path consistency, path hit rate, and health-oriented plausibility. These findings indicate that risk-aware causal path re-ranking improved risk ranking and explanation grounding relative to the evaluated baselines under the controlled benchmark setting.</description>
	<pubDate>2026-07-28</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 248: RACPR: Generative AI-Enhanced Risk-Aware Causal Path Re-Ranking for Interpretable Multimorbidity Risk Identification in Elderly Health Consultation Scenarios</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/8/248">doi: 10.3390/bdcc10080248</a></p>
	<p>Authors:
		Shaofu Lin
		Shaojie Wang
		Zhisheng Huang
		Haoru Su
		</p>
	<p>Multimorbidity risk identification in elderly health consultation scenarios is challenging because chronic disease history and medication exposure may interact through complex causal pathways. Existing LLM-, RAG-, and graph-based retrieval methods often rely on semantic relevance or graph connectivity, which may be insufficient for identifying user-specific risk mechanisms. This study proposes RACPR, a generative AI-enhanced risk-aware causal path re-ranking framework for interpretable multimorbidity risk identification. RACPR ranks candidate causal paths by integrating user-entity alignment, causal coherence, risk contribution, and path length control and uses generative AI to transform selected paths into readable, path-grounded explanations. To support controlled algorithmic evaluation, we constructed a normalized benchmark of 1002 elderly multimorbidity consultation cases covering diabetes, hypertension, and chronic kidney disease. The benchmark and supporting knowledge graph were derived from publicly available biomedical and health information resources, including PubMed abstracts, guideline and review sources, DrugBank medication-safety evidence, and MedlinePlus-based terminology. Under a leakage-controlled setting, only age, diagnosed diseases, and medication exposures were used as model-accessible inputs, while abnormal indicators, support paths, and rationales were reserved for evaluation. On the test set, RACPR achieved an Accuracy of 0.659, a Precision of 0.602, a Recall of 0.938, an F1-score of 0.733, and an AUC of 0.777. Ablation analysis showed that removing the risk-aware component reduced AUC from 0.777 to 0.428. Benchmark-level explanation evaluation further showed improvements in path consistency, path hit rate, and health-oriented plausibility. These findings indicate that risk-aware causal path re-ranking improved risk ranking and explanation grounding relative to the evaluated baselines under the controlled benchmark setting.</p>
	]]></content:encoded>

	<dc:title>RACPR: Generative AI-Enhanced Risk-Aware Causal Path Re-Ranking for Interpretable Multimorbidity Risk Identification in Elderly Health Consultation Scenarios</dc:title>
			<dc:creator>Shaofu Lin</dc:creator>
			<dc:creator>Shaojie Wang</dc:creator>
			<dc:creator>Zhisheng Huang</dc:creator>
			<dc:creator>Haoru Su</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10080248</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-07-28</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-07-28</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>8</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>248</prism:startingPage>
		<prism:doi>10.3390/bdcc10080248</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/8/248</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/8/247">

	<title>BDCC, Vol. 10, Pages 247: Graph-Regularized Low-Rank Label Correlation Learning with Label-Specific Features for Missing Labels</title>
	<link>https://www.mdpi.com/2504-2289/10/8/247</link>
	<description>Missing labels are common in multi-label learning and can bias both label-correlation estimation and classifier induction. Existing missing-label methods often recover incomplete supervision mainly through global label correlations. However, correlation-driven recovery alone may produce over-smoothed supervision when annotations are sparse, while label-specific discriminative evidence may be weakened. To address this problem, we propose GLCS, a graph-regularized low-rank correlation learning framework with label-specific features for multi-label learning with missing labels. GLCS first uses the observed entries as reliable supervision sources and propagates them through a learned label correlation matrix. It then jointly learns sparse label-specific predictors, low-rank label correlations, and a label graph regularizer induced by the learned correlations. In this way, global label dependencies, local label-structure consistency, and label-wise discriminative features are optimized in a unified objective. The resulting problem is solved by an alternating proximal optimization scheme with soft thresholding for sparse predictors and singular value thresholding for low-rank correlations. Experiments on twelve benchmark datasets under three missing-label ratios show that GLCS obtains strong average performance across AP, AUC, CV, HL, OE, and RL, especially under high missing rates.</description>
	<pubDate>2026-07-27</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 247: Graph-Regularized Low-Rank Label Correlation Learning with Label-Specific Features for Missing Labels</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/8/247">doi: 10.3390/bdcc10080247</a></p>
	<p>Authors:
		Tianli Li
		Mohammad Faidzul Nasrudin
		Xing Peng
		Huifen Xing
		</p>
	<p>Missing labels are common in multi-label learning and can bias both label-correlation estimation and classifier induction. Existing missing-label methods often recover incomplete supervision mainly through global label correlations. However, correlation-driven recovery alone may produce over-smoothed supervision when annotations are sparse, while label-specific discriminative evidence may be weakened. To address this problem, we propose GLCS, a graph-regularized low-rank correlation learning framework with label-specific features for multi-label learning with missing labels. GLCS first uses the observed entries as reliable supervision sources and propagates them through a learned label correlation matrix. It then jointly learns sparse label-specific predictors, low-rank label correlations, and a label graph regularizer induced by the learned correlations. In this way, global label dependencies, local label-structure consistency, and label-wise discriminative features are optimized in a unified objective. The resulting problem is solved by an alternating proximal optimization scheme with soft thresholding for sparse predictors and singular value thresholding for low-rank correlations. Experiments on twelve benchmark datasets under three missing-label ratios show that GLCS obtains strong average performance across AP, AUC, CV, HL, OE, and RL, especially under high missing rates.</p>
	]]></content:encoded>

	<dc:title>Graph-Regularized Low-Rank Label Correlation Learning with Label-Specific Features for Missing Labels</dc:title>
			<dc:creator>Tianli Li</dc:creator>
			<dc:creator>Mohammad Faidzul Nasrudin</dc:creator>
			<dc:creator>Xing Peng</dc:creator>
			<dc:creator>Huifen Xing</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10080247</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-07-27</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-07-27</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>8</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>247</prism:startingPage>
		<prism:doi>10.3390/bdcc10080247</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/8/247</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/8/246">

	<title>BDCC, Vol. 10, Pages 246: Visual Safe Human-to-Humanoid Motion Imitation</title>
	<link>https://www.mdpi.com/2504-2289/10/8/246</link>
	<description>Safe human-to-humanoid motion imitation is crucial for shared environments, where direct motion retargeting may induce self-collision or human&amp;amp;ndash;humanoid collision due to embodiment mismatch, kinematic limits, perception uncertainty, and human proximity. This paper presents an online vision-aided safe human-to-humanoid motion imitation framework that integrates skeleton-based upper-body pose estimation, joint-space retargeting, and capsule-based Control Barrier Function Quadratic Program (CBF-QP) safety filtering. Human skeletal observations are mapped to a reduced eight-degree-of-freedom (8-DoF) humanoid upper-body command, while the CBF-QP layer computes a safety-corrected target that minimally modifies the nominal imitation command subject to robot self-collision and human&amp;amp;ndash;humanoid collision constraints. The framework is evaluated via simulation and hardware experiments under representative self-collision and human&amp;amp;ndash;humanoid interaction episodes, complemented by a comparative benchmark against velocity damping and potential field baselines. Furthermore, this work introduces an evaluation protocol combining geometric safety, command deviation, and local-link similarity metrics to systematically characterize the safety&amp;amp;ndash;imitation trade-off governed by CBF parameters. The results demonstrate that, within the tested moderate-speed regime, the proposed framework substantially reduces geometric collision violations while balancing imitation fidelity with online computational feasibility, thereby providing a viable foundation for safe human-guided humanoid motion deployment.</description>
	<pubDate>2026-07-24</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 246: Visual Safe Human-to-Humanoid Motion Imitation</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/8/246">doi: 10.3390/bdcc10080246</a></p>
	<p>Authors:
		Wenqi Cai
		John Abanes
		Nikolaos Evangeliou
		Anthony Tzes
		</p>
	<p>Safe human-to-humanoid motion imitation is crucial for shared environments, where direct motion retargeting may induce self-collision or human&amp;amp;ndash;humanoid collision due to embodiment mismatch, kinematic limits, perception uncertainty, and human proximity. This paper presents an online vision-aided safe human-to-humanoid motion imitation framework that integrates skeleton-based upper-body pose estimation, joint-space retargeting, and capsule-based Control Barrier Function Quadratic Program (CBF-QP) safety filtering. Human skeletal observations are mapped to a reduced eight-degree-of-freedom (8-DoF) humanoid upper-body command, while the CBF-QP layer computes a safety-corrected target that minimally modifies the nominal imitation command subject to robot self-collision and human&amp;amp;ndash;humanoid collision constraints. The framework is evaluated via simulation and hardware experiments under representative self-collision and human&amp;amp;ndash;humanoid interaction episodes, complemented by a comparative benchmark against velocity damping and potential field baselines. Furthermore, this work introduces an evaluation protocol combining geometric safety, command deviation, and local-link similarity metrics to systematically characterize the safety&amp;amp;ndash;imitation trade-off governed by CBF parameters. The results demonstrate that, within the tested moderate-speed regime, the proposed framework substantially reduces geometric collision violations while balancing imitation fidelity with online computational feasibility, thereby providing a viable foundation for safe human-guided humanoid motion deployment.</p>
	]]></content:encoded>

	<dc:title>Visual Safe Human-to-Humanoid Motion Imitation</dc:title>
			<dc:creator>Wenqi Cai</dc:creator>
			<dc:creator>John Abanes</dc:creator>
			<dc:creator>Nikolaos Evangeliou</dc:creator>
			<dc:creator>Anthony Tzes</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10080246</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-07-24</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-07-24</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>8</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>246</prism:startingPage>
		<prism:doi>10.3390/bdcc10080246</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/8/246</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/8/245">

	<title>BDCC, Vol. 10, Pages 245: Accelerated Energy Forecasting Models with Metaheuristics for Swift Solutions in Building Management</title>
	<link>https://www.mdpi.com/2504-2289/10/8/245</link>
	<description>This work presents a CUDA-accelerated methodology for training multiple neural networks in parallel using population-based metaheuristics. The goal is to obtain fast and accurate short-term energy-forecasting models for time-sensitive building-management applications. We evaluate five metaheuristic optimizers and their memetic variants, for which a local-search stage based on the ADAM optimizer is incorporated. The models are assessed on eleven real-world energy-consumption time series using training time, root mean squared error (RMSE), mean absolute error (MAE), and normalized RMSE (NRMSE). The results show that the proposed memetic approaches are competitive under strict training-time budgets, although unconstrained ADAM remains a strong overall reference.</description>
	<pubDate>2026-07-23</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 245: Accelerated Energy Forecasting Models with Metaheuristics for Swift Solutions in Building Management</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/8/245">doi: 10.3390/bdcc10080245</a></p>
	<p>Authors:
		D. Criado-Ramón
		Maria Ruxandra Cojocaru
		L. G. B. Ruiz
		Lorenzo Servadei
		Robert Wille
		M. P. Cuéllar
		M. C. Pegalajar
		</p>
	<p>This work presents a CUDA-accelerated methodology for training multiple neural networks in parallel using population-based metaheuristics. The goal is to obtain fast and accurate short-term energy-forecasting models for time-sensitive building-management applications. We evaluate five metaheuristic optimizers and their memetic variants, for which a local-search stage based on the ADAM optimizer is incorporated. The models are assessed on eleven real-world energy-consumption time series using training time, root mean squared error (RMSE), mean absolute error (MAE), and normalized RMSE (NRMSE). The results show that the proposed memetic approaches are competitive under strict training-time budgets, although unconstrained ADAM remains a strong overall reference.</p>
	]]></content:encoded>

	<dc:title>Accelerated Energy Forecasting Models with Metaheuristics for Swift Solutions in Building Management</dc:title>
			<dc:creator>D. Criado-Ramón</dc:creator>
			<dc:creator>Maria Ruxandra Cojocaru</dc:creator>
			<dc:creator>L. G. B. Ruiz</dc:creator>
			<dc:creator>Lorenzo Servadei</dc:creator>
			<dc:creator>Robert Wille</dc:creator>
			<dc:creator>M. P. Cuéllar</dc:creator>
			<dc:creator>M. C. Pegalajar</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10080245</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-07-23</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-07-23</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>8</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>245</prism:startingPage>
		<prism:doi>10.3390/bdcc10080245</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/8/245</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/244">

	<title>BDCC, Vol. 10, Pages 244: Safety-Aware Event-Triggered Intervention for Motion Planning and Decision Making in Diffusion-Based Autonomous Driving</title>
	<link>https://www.mdpi.com/2504-2289/10/7/244</link>
	<description>Diffusion-based trajectory planners achieve strong nominal performance in autonomous driving, but sparse safety intervention remains difficult to evaluate and realize effectively. This study addresses this problem by proposing a safety-aware event-triggered intervention framework on top of a fixed DiffusionDriveV2 planner. The method uses candidate-level risk signals and an auxiliary semantic risk trigger to decide when intervention should be activated, and realizes the intervention through conservative re-selection and mild action-space augmentation. To evaluate sparse interventions beyond global validation metrics, we further construct normal, conservative, and actual trajectories and introduce triggered-subset counterfactual evaluation. On NAVSIM navtest, global planner metrics remain nearly unchanged across sparse trigger policies, but the semantic trigger achieves better triggered-subset final score, TTC, and progress than matched-random and TTC-based triggers. Qualitative cases show that behaviorally distinct safety responses mainly arise from action-space augmentation rather than candidate reranking alone. These results show that the proposed framework can diagnose and partially alleviate the gap between risk recognition and action realization, while revealing that stronger semantic-conditioned action generation is needed to fully overcome the trigger-to-action bottleneck.</description>
	<pubDate>2026-07-21</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 244: Safety-Aware Event-Triggered Intervention for Motion Planning and Decision Making in Diffusion-Based Autonomous Driving</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/244">doi: 10.3390/bdcc10070244</a></p>
	<p>Authors:
		Xuerui Fang
		Hui Li
		Zehao Xue
		Gulimila Kezierbieke
		</p>
	<p>Diffusion-based trajectory planners achieve strong nominal performance in autonomous driving, but sparse safety intervention remains difficult to evaluate and realize effectively. This study addresses this problem by proposing a safety-aware event-triggered intervention framework on top of a fixed DiffusionDriveV2 planner. The method uses candidate-level risk signals and an auxiliary semantic risk trigger to decide when intervention should be activated, and realizes the intervention through conservative re-selection and mild action-space augmentation. To evaluate sparse interventions beyond global validation metrics, we further construct normal, conservative, and actual trajectories and introduce triggered-subset counterfactual evaluation. On NAVSIM navtest, global planner metrics remain nearly unchanged across sparse trigger policies, but the semantic trigger achieves better triggered-subset final score, TTC, and progress than matched-random and TTC-based triggers. Qualitative cases show that behaviorally distinct safety responses mainly arise from action-space augmentation rather than candidate reranking alone. These results show that the proposed framework can diagnose and partially alleviate the gap between risk recognition and action realization, while revealing that stronger semantic-conditioned action generation is needed to fully overcome the trigger-to-action bottleneck.</p>
	]]></content:encoded>

	<dc:title>Safety-Aware Event-Triggered Intervention for Motion Planning and Decision Making in Diffusion-Based Autonomous Driving</dc:title>
			<dc:creator>Xuerui Fang</dc:creator>
			<dc:creator>Hui Li</dc:creator>
			<dc:creator>Zehao Xue</dc:creator>
			<dc:creator>Gulimila Kezierbieke</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070244</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-07-21</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-07-21</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>244</prism:startingPage>
		<prism:doi>10.3390/bdcc10070244</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/244</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/243">

	<title>BDCC, Vol. 10, Pages 243: ARAS-H-IW: A Reproducible Hesitant Fuzzy Multi-Criteria Decision Framework with Inverse Weight Inference for Healthcare Waste Treatment Technology Assessment</title>
	<link>https://www.mdpi.com/2504-2289/10/7/243</link>
	<description>Selecting appropriate healthcare waste (HW) treatment technologies is a challenging multi-criteria decision-making problem characterized by uncertainty, conflicting evaluation criteria, and limited decision-support information. Existing approaches often rely on subjective weighting schemes and may provide rankings that are sensitive to variations in expert judgments. This study proposes ARAS-H-IW, a hybrid decision-support framework that combines Additive Ratio Assessment under Hesitant Fuzzy Sets (ARAS-H) with an Inverse Weighting (IW) mechanism capable of inferring criterion weights directly from expert preference constraints through constrained quadratic optimization. To evaluate its practical applicability, the framework was applied to a real-world healthcare waste management case study using data provided by the Regional Health Directorate of Fez-Meknes (Morocco). A fully reproducible Python-based web platform was developed to automate the complete analytical workflow, including hesitant fuzzy modeling, multi-expert ranking aggregation, inverse weight inference, comparative evaluation, sensitivity analysis, Monte Carlo robustness assessment, and automated reporting. The proposed framework identified centralized autoclaving as the most favorable treatment alternative, followed by regional outsourcing and microwave disinfection. Comparative analyses with TOPSIS, VIKOR, PROMETHEE II, and EDAS showed strong agreement regarding the best- and worst-ranked alternatives. Sensitivity and Monte Carlo analyses further demonstrated the stability and robustness of the obtained rankings, while all expert aggregation strategies converged toward the same consensus ordering. The results highlight the capacity of ARAS-H-IW to generate transparent, reproducible, and robust decision recommendations under uncertainty. The proposed framework provides a practical tool for healthcare waste technology assessment and offers a promising foundation for supporting evidence-based decision-making in regional healthcare waste management.</description>
	<pubDate>2026-07-18</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 243: ARAS-H-IW: A Reproducible Hesitant Fuzzy Multi-Criteria Decision Framework with Inverse Weight Inference for Healthcare Waste Treatment Technology Assessment</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/243">doi: 10.3390/bdcc10070243</a></p>
	<p>Authors:
		Hassnae Aberkane
		Latifa Boubekri
		Hafid El Karimi
		Mohammed Chaouki Abounaima
		</p>
	<p>Selecting appropriate healthcare waste (HW) treatment technologies is a challenging multi-criteria decision-making problem characterized by uncertainty, conflicting evaluation criteria, and limited decision-support information. Existing approaches often rely on subjective weighting schemes and may provide rankings that are sensitive to variations in expert judgments. This study proposes ARAS-H-IW, a hybrid decision-support framework that combines Additive Ratio Assessment under Hesitant Fuzzy Sets (ARAS-H) with an Inverse Weighting (IW) mechanism capable of inferring criterion weights directly from expert preference constraints through constrained quadratic optimization. To evaluate its practical applicability, the framework was applied to a real-world healthcare waste management case study using data provided by the Regional Health Directorate of Fez-Meknes (Morocco). A fully reproducible Python-based web platform was developed to automate the complete analytical workflow, including hesitant fuzzy modeling, multi-expert ranking aggregation, inverse weight inference, comparative evaluation, sensitivity analysis, Monte Carlo robustness assessment, and automated reporting. The proposed framework identified centralized autoclaving as the most favorable treatment alternative, followed by regional outsourcing and microwave disinfection. Comparative analyses with TOPSIS, VIKOR, PROMETHEE II, and EDAS showed strong agreement regarding the best- and worst-ranked alternatives. Sensitivity and Monte Carlo analyses further demonstrated the stability and robustness of the obtained rankings, while all expert aggregation strategies converged toward the same consensus ordering. The results highlight the capacity of ARAS-H-IW to generate transparent, reproducible, and robust decision recommendations under uncertainty. The proposed framework provides a practical tool for healthcare waste technology assessment and offers a promising foundation for supporting evidence-based decision-making in regional healthcare waste management.</p>
	]]></content:encoded>

	<dc:title>ARAS-H-IW: A Reproducible Hesitant Fuzzy Multi-Criteria Decision Framework with Inverse Weight Inference for Healthcare Waste Treatment Technology Assessment</dc:title>
			<dc:creator>Hassnae Aberkane</dc:creator>
			<dc:creator>Latifa Boubekri</dc:creator>
			<dc:creator>Hafid El Karimi</dc:creator>
			<dc:creator>Mohammed Chaouki Abounaima</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070243</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-07-18</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-07-18</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>243</prism:startingPage>
		<prism:doi>10.3390/bdcc10070243</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/243</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/242">

	<title>BDCC, Vol. 10, Pages 242: MSFA: Multi-Strategy Fusion Algorithm for Data Cleaning and Its Application in Offshore Marine Environmental Monitoring</title>
	<link>https://www.mdpi.com/2504-2289/10/7/242</link>
	<description>Marine monitoring records collected from buoys and nearshore sensors are often affected by missing values, abrupt spikes, and short-term fluctuations. These errors are difficult to remove with a single detection or interpolation rule, especially when local anomalies and global outliers occur in the same sequence. This study develops a Multi-Strategy Fusion Architecture (MSFA) for cleaning marine environmental time-series data. In MSFA, DBSCAN is not applied directly to the raw observations; instead, the time index and measurement value are first normalized into a common feature space, where local density anomalies can be detected more consistently. IQR screening is then used to identify global extreme values. After abnormal positions are marked, the repair result is estimated from two complementary sources: linear interpolation, which follows local temporal change, and a moving average based only on neighboring valid observations, which reduces random noise. Their contributions are adjusted according to local reliability rather than fixed manually. Because initial repair may still leave small residual errors, we further use a Combined Residual Metric (CRM) with a median/MAD-based threshold to recheck the repaired sequence and update the abnormal-position set when necessary. Experiments on the 2020 Dongying offshore buoy dataset and a self-collected nearshore dataset show that MSFA achieves AUROC/AUPRC/NRMSE values of 0.896/0.855/0.066 and 0.986/0.915/0.0653, respectively. Compared with DBSCAN+LOF, DBSCAN+Transformer, and IQR+Sigmoid, MSFA improves AUROC and AUPRC by about 12&amp;amp;ndash;25% on average and reduces NRMSE by more than 40%. These results indicate that the proposed method can improve the usability of noisy and incomplete marine monitoring data while keeping the cleaning process interpretable.</description>
	<pubDate>2026-07-17</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 242: MSFA: Multi-Strategy Fusion Algorithm for Data Cleaning and Its Application in Offshore Marine Environmental Monitoring</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/242">doi: 10.3390/bdcc10070242</a></p>
	<p>Authors:
		Kun Chen
		Ruikang Chang
		Li Ma
		Chao Ji
		Quan Liu
		</p>
	<p>Marine monitoring records collected from buoys and nearshore sensors are often affected by missing values, abrupt spikes, and short-term fluctuations. These errors are difficult to remove with a single detection or interpolation rule, especially when local anomalies and global outliers occur in the same sequence. This study develops a Multi-Strategy Fusion Architecture (MSFA) for cleaning marine environmental time-series data. In MSFA, DBSCAN is not applied directly to the raw observations; instead, the time index and measurement value are first normalized into a common feature space, where local density anomalies can be detected more consistently. IQR screening is then used to identify global extreme values. After abnormal positions are marked, the repair result is estimated from two complementary sources: linear interpolation, which follows local temporal change, and a moving average based only on neighboring valid observations, which reduces random noise. Their contributions are adjusted according to local reliability rather than fixed manually. Because initial repair may still leave small residual errors, we further use a Combined Residual Metric (CRM) with a median/MAD-based threshold to recheck the repaired sequence and update the abnormal-position set when necessary. Experiments on the 2020 Dongying offshore buoy dataset and a self-collected nearshore dataset show that MSFA achieves AUROC/AUPRC/NRMSE values of 0.896/0.855/0.066 and 0.986/0.915/0.0653, respectively. Compared with DBSCAN+LOF, DBSCAN+Transformer, and IQR+Sigmoid, MSFA improves AUROC and AUPRC by about 12&amp;amp;ndash;25% on average and reduces NRMSE by more than 40%. These results indicate that the proposed method can improve the usability of noisy and incomplete marine monitoring data while keeping the cleaning process interpretable.</p>
	]]></content:encoded>

	<dc:title>MSFA: Multi-Strategy Fusion Algorithm for Data Cleaning and Its Application in Offshore Marine Environmental Monitoring</dc:title>
			<dc:creator>Kun Chen</dc:creator>
			<dc:creator>Ruikang Chang</dc:creator>
			<dc:creator>Li Ma</dc:creator>
			<dc:creator>Chao Ji</dc:creator>
			<dc:creator>Quan Liu</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070242</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-07-17</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-07-17</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>242</prism:startingPage>
		<prism:doi>10.3390/bdcc10070242</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/242</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/241">

	<title>BDCC, Vol. 10, Pages 241: A Model for Metadata Organisation and Management for Compilation of Specialised Datasets from Big Data</title>
	<link>https://www.mdpi.com/2504-2289/10/7/241</link>
	<description>The paper presents a model for the design and management of metadata that enables the efficient compilation of specialised datasets from large, heterogeneous data collections. The metadata are represented as a typed property graph that facilitates the FAIR principles in data compilation: Findable, Accessible, Interoperable, Reusable. The representation is general and independent of the modality and format of the data. The graph-based design of the metadata supports the incremental extension of categories and relationships without requiring the migration of existing data. Its feasibility is demonstrated through the use of a graph database, in which the metadata for 689,645 Bulgarian textual data units are combined with a web-based filtering interface. Metadata retrieval is implemented through Cypher queries executed as graph traversals, enabling the extraction of thematic and application-oriented data subsets based on combinations of selection criteria. The application validates the suitability of the graph-based metadata design for compiling specialised datasets for training and fine-tuning large language models and other NLP applications.</description>
	<pubDate>2026-07-17</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 241: A Model for Metadata Organisation and Management for Compilation of Specialised Datasets from Big Data</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/241">doi: 10.3390/bdcc10070241</a></p>
	<p>Authors:
		Svetla Koeva
		Ivelina Stoyanova
		</p>
	<p>The paper presents a model for the design and management of metadata that enables the efficient compilation of specialised datasets from large, heterogeneous data collections. The metadata are represented as a typed property graph that facilitates the FAIR principles in data compilation: Findable, Accessible, Interoperable, Reusable. The representation is general and independent of the modality and format of the data. The graph-based design of the metadata supports the incremental extension of categories and relationships without requiring the migration of existing data. Its feasibility is demonstrated through the use of a graph database, in which the metadata for 689,645 Bulgarian textual data units are combined with a web-based filtering interface. Metadata retrieval is implemented through Cypher queries executed as graph traversals, enabling the extraction of thematic and application-oriented data subsets based on combinations of selection criteria. The application validates the suitability of the graph-based metadata design for compiling specialised datasets for training and fine-tuning large language models and other NLP applications.</p>
	]]></content:encoded>

	<dc:title>A Model for Metadata Organisation and Management for Compilation of Specialised Datasets from Big Data</dc:title>
			<dc:creator>Svetla Koeva</dc:creator>
			<dc:creator>Ivelina Stoyanova</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070241</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-07-17</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-07-17</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>241</prism:startingPage>
		<prism:doi>10.3390/bdcc10070241</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/241</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/240">

	<title>BDCC, Vol. 10, Pages 240: Cognitive Big Data Architecture for Daily Operational Jamming Transition Detection with Low-Latency Inference in Infrastructure-Constrained Financial Markets: The MERI Framework</title>
	<link>https://www.mdpi.com/2504-2289/10/7/240</link>
	<description>We introduce the MERI (Market Evolutionary Resilience Index), a cognitive big data framework operationalising jamming transition physics into a daily operational regime detector (low-latency inference, 8 ms per observation). Opaque models cannot be deployed in regulated environments because every automated alert must decompose into auditable feature contributions. The MERI addresses this by treating the market as a complex adaptive system whose metabolic state constitutes the primary observable. Three cognitive layers fuse heterogeneous streaming data: an EGARCH-GED econometric baseline, a Random Forest classifier on a 15-dimensional physics-derived feature space, and a TreeSHAP Gini attribution audit ensuring full prediction-level transparency. Fisher Information Gain epistemic gating restricts automated intervention to predictions exceeding 2.5 nats certainty. Evaluated on South African financial markets (2015&amp;amp;ndash;2025, N=2870 trading days, Eskom load-shedding as exogenous forcing), the MERI achieves 97.3% accuracy (AUC = 0.9973, recall = 1.000), statistically equivalent to Temporal Fusion Transformers (Model Confidence Set, 90% confidence) while delivering 85.7% high-certainty predictions versus 23.4% for deep learning. A Granger-validated 48-h early warning lead (F=62.003, p&amp;amp;lt;0.001), 7.78&amp;amp;times; recovery hysteresis (Cohen&amp;amp;rsquo;s d=2.13), and infrastructure dominance of 78.0% (Gini) confirm that the framework is operationally feasible for daily monitoring in the South African JSE&amp;amp;ndash;Eskom setting. Cross-domain portability is proposed as a theoretical extension pending empirical validation.</description>
	<pubDate>2026-07-16</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 240: Cognitive Big Data Architecture for Daily Operational Jamming Transition Detection with Low-Latency Inference in Infrastructure-Constrained Financial Markets: The MERI Framework</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/240">doi: 10.3390/bdcc10070240</a></p>
	<p>Authors:
		Ntebogang Dinah Moroke
		</p>
	<p>We introduce the MERI (Market Evolutionary Resilience Index), a cognitive big data framework operationalising jamming transition physics into a daily operational regime detector (low-latency inference, 8 ms per observation). Opaque models cannot be deployed in regulated environments because every automated alert must decompose into auditable feature contributions. The MERI addresses this by treating the market as a complex adaptive system whose metabolic state constitutes the primary observable. Three cognitive layers fuse heterogeneous streaming data: an EGARCH-GED econometric baseline, a Random Forest classifier on a 15-dimensional physics-derived feature space, and a TreeSHAP Gini attribution audit ensuring full prediction-level transparency. Fisher Information Gain epistemic gating restricts automated intervention to predictions exceeding 2.5 nats certainty. Evaluated on South African financial markets (2015&amp;amp;ndash;2025, N=2870 trading days, Eskom load-shedding as exogenous forcing), the MERI achieves 97.3% accuracy (AUC = 0.9973, recall = 1.000), statistically equivalent to Temporal Fusion Transformers (Model Confidence Set, 90% confidence) while delivering 85.7% high-certainty predictions versus 23.4% for deep learning. A Granger-validated 48-h early warning lead (F=62.003, p&amp;amp;lt;0.001), 7.78&amp;amp;times; recovery hysteresis (Cohen&amp;amp;rsquo;s d=2.13), and infrastructure dominance of 78.0% (Gini) confirm that the framework is operationally feasible for daily monitoring in the South African JSE&amp;amp;ndash;Eskom setting. Cross-domain portability is proposed as a theoretical extension pending empirical validation.</p>
	]]></content:encoded>

	<dc:title>Cognitive Big Data Architecture for Daily Operational Jamming Transition Detection with Low-Latency Inference in Infrastructure-Constrained Financial Markets: The MERI Framework</dc:title>
			<dc:creator>Ntebogang Dinah Moroke</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070240</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-07-16</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-07-16</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>240</prism:startingPage>
		<prism:doi>10.3390/bdcc10070240</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/240</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/239">

	<title>BDCC, Vol. 10, Pages 239: Cognitive Detection at Big-Data Scale: A CNN-LSTM-DQN Framework with Prioritized Experience Replay for Cross-Attack-Family Generalization and Multi-Seed Initialization Sensitivity Analysis</title>
	<link>https://www.mdpi.com/2504-2289/10/7/239</link>
	<description>Real-world IoT network security generates traffic at big-data scale with extreme class imbalance, temporal non-stationarity, and continuously evolving attack strategies that overwhelm static supervised classifiers. This paper presents a cognitive computing framework for network intrusion detection: a CNN&amp;amp;ndash;LSTM&amp;amp;ndash;DQN architecture with Prioritized Experience Replay (PER) evaluated on a 5,000,000-flow naturalistic sample of the TON_IoT Processed_Network dataset (4,000,000 training/1,000,000 temporally held-out test flows; 94.5% attack ratio) under a strict temporal split. The cognitive agent optimizes detection decisions using an Alerts per Million Flows (ARMF)-aware reward function that encodes both alert-fatigue cost and missed-attack penalty. We conduct a cross-attack-family generalization study: the methodology&amp;amp;mdash;architecture template, reward design, and hyperparameter calibration&amp;amp;mdash;is inherited from a framework previously validated on CSE-CIC-IDS2018, re-instantiated and retrained on the structurally different TON_IoT environment, and compared against the previously published benchmark. Initialization sensitivity is characterized across five independent random seeds using paired Wilcoxon signed-rank and t-tests. Across the five seeds, the proposed X2 model attains recall 0.833 &amp;amp;plusmn; 0.306 and F1 0.874 &amp;amp;plusmn; 0.241 (mean &amp;amp;plusmn; sample SD), versus the supervised X1 baseline at 0.858 &amp;amp;plusmn; 0.178 and 0.912 &amp;amp;plusmn; 0.116; the best-performing seed (42) achieves 97.52% accuracy, 98.02% attack recall, 99.46% precision, and 98.73% F1-score on 1,000,000 held-out XSS flows&amp;amp;mdash;an attack family entirely absent from training&amp;amp;mdash;with temporal stability variances of 4.63 &amp;amp;times; 10&amp;amp;minus;7 (recall) and 1.38 &amp;amp;times; 10&amp;amp;minus;7 (F1). The X2 advantage observed among the four stable seeds is not statistically demonstrated at n = 5 (statistical power &amp;amp;asymp; 5.1%); the initialization-sensitivity finding itself, including one degenerate alert-suppression seed, is reported as a primary contribution. A formal, exactly additive ARMF decomposition distinguishes the detected-attack (structural) component (99.46%) from the model-induced false-positive component (0.54%), and we report a multi-seed, ARMF-aware cognitive IDS evaluation on naturalistic TON_IoT traffic under an unseen-attack-family test condition that, to the best of our knowledge, has not been reported in the surveyed RL-based NIDS literature.</description>
	<pubDate>2026-07-16</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 239: Cognitive Detection at Big-Data Scale: A CNN-LSTM-DQN Framework with Prioritized Experience Replay for Cross-Attack-Family Generalization and Multi-Seed Initialization Sensitivity Analysis</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/239">doi: 10.3390/bdcc10070239</a></p>
	<p>Authors:
		 Rushendra
		Kalamullah Ramli
		Prima Dewi Purnamasari
		Teddy Surya Gunawan
		Muhammad Salman
		</p>
	<p>Real-world IoT network security generates traffic at big-data scale with extreme class imbalance, temporal non-stationarity, and continuously evolving attack strategies that overwhelm static supervised classifiers. This paper presents a cognitive computing framework for network intrusion detection: a CNN&amp;amp;ndash;LSTM&amp;amp;ndash;DQN architecture with Prioritized Experience Replay (PER) evaluated on a 5,000,000-flow naturalistic sample of the TON_IoT Processed_Network dataset (4,000,000 training/1,000,000 temporally held-out test flows; 94.5% attack ratio) under a strict temporal split. The cognitive agent optimizes detection decisions using an Alerts per Million Flows (ARMF)-aware reward function that encodes both alert-fatigue cost and missed-attack penalty. We conduct a cross-attack-family generalization study: the methodology&amp;amp;mdash;architecture template, reward design, and hyperparameter calibration&amp;amp;mdash;is inherited from a framework previously validated on CSE-CIC-IDS2018, re-instantiated and retrained on the structurally different TON_IoT environment, and compared against the previously published benchmark. Initialization sensitivity is characterized across five independent random seeds using paired Wilcoxon signed-rank and t-tests. Across the five seeds, the proposed X2 model attains recall 0.833 &amp;amp;plusmn; 0.306 and F1 0.874 &amp;amp;plusmn; 0.241 (mean &amp;amp;plusmn; sample SD), versus the supervised X1 baseline at 0.858 &amp;amp;plusmn; 0.178 and 0.912 &amp;amp;plusmn; 0.116; the best-performing seed (42) achieves 97.52% accuracy, 98.02% attack recall, 99.46% precision, and 98.73% F1-score on 1,000,000 held-out XSS flows&amp;amp;mdash;an attack family entirely absent from training&amp;amp;mdash;with temporal stability variances of 4.63 &amp;amp;times; 10&amp;amp;minus;7 (recall) and 1.38 &amp;amp;times; 10&amp;amp;minus;7 (F1). The X2 advantage observed among the four stable seeds is not statistically demonstrated at n = 5 (statistical power &amp;amp;asymp; 5.1%); the initialization-sensitivity finding itself, including one degenerate alert-suppression seed, is reported as a primary contribution. A formal, exactly additive ARMF decomposition distinguishes the detected-attack (structural) component (99.46%) from the model-induced false-positive component (0.54%), and we report a multi-seed, ARMF-aware cognitive IDS evaluation on naturalistic TON_IoT traffic under an unseen-attack-family test condition that, to the best of our knowledge, has not been reported in the surveyed RL-based NIDS literature.</p>
	]]></content:encoded>

	<dc:title>Cognitive Detection at Big-Data Scale: A CNN-LSTM-DQN Framework with Prioritized Experience Replay for Cross-Attack-Family Generalization and Multi-Seed Initialization Sensitivity Analysis</dc:title>
			<dc:creator> Rushendra</dc:creator>
			<dc:creator>Kalamullah Ramli</dc:creator>
			<dc:creator>Prima Dewi Purnamasari</dc:creator>
			<dc:creator>Teddy Surya Gunawan</dc:creator>
			<dc:creator>Muhammad Salman</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070239</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-07-16</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-07-16</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>239</prism:startingPage>
		<prism:doi>10.3390/bdcc10070239</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/239</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/238">

	<title>BDCC, Vol. 10, Pages 238: Comparative Keyword Network Analysis of Korean-Language Algorithmic Recommendation Discourses in AI Related to TikTok and YouTube</title>
	<link>https://www.mdpi.com/2504-2289/10/7/238</link>
	<description>This study investigates how artificial intelligence (AI) is represented within Korean-language recommendation algorithm discourse on TikTok and YouTube. To examine the structural characteristics and discourse tendencies of AI-related discussions, the study applies text mining and keyword network analysis methods, including TF, TF-IDF analysis, centrality analysis, CONCOR clustering, and sentiment analysis. The findings indicate that AI occupies a central position within recommendation algorithm discourse and is strongly associated with algorithms, data, content recommendation, and technological systems across both platforms. The analysis further reveals notable differences between the two platforms: TikTok discourse demonstrates a stronger emphasis on automation and technological mechanisms, whereas YouTube discourse is more closely associated with content production, commercialization, and educational contexts. In addition, public discourse surrounding AI-driven recommendation systems reflects both positive-oriented and concern-related perspectives regarding technological innovation, platform influence, and social implications. This study contributes to a broader understanding of how AI is socially interpreted and represented within contemporary digital platform discourse.</description>
	<pubDate>2026-07-16</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 238: Comparative Keyword Network Analysis of Korean-Language Algorithmic Recommendation Discourses in AI Related to TikTok and YouTube</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/238">doi: 10.3390/bdcc10070238</a></p>
	<p>Authors:
		Dae Wan Kim
		Luman Dong
		Guihua Zhang
		Xu Yin
		Yujong Hwang
		</p>
	<p>This study investigates how artificial intelligence (AI) is represented within Korean-language recommendation algorithm discourse on TikTok and YouTube. To examine the structural characteristics and discourse tendencies of AI-related discussions, the study applies text mining and keyword network analysis methods, including TF, TF-IDF analysis, centrality analysis, CONCOR clustering, and sentiment analysis. The findings indicate that AI occupies a central position within recommendation algorithm discourse and is strongly associated with algorithms, data, content recommendation, and technological systems across both platforms. The analysis further reveals notable differences between the two platforms: TikTok discourse demonstrates a stronger emphasis on automation and technological mechanisms, whereas YouTube discourse is more closely associated with content production, commercialization, and educational contexts. In addition, public discourse surrounding AI-driven recommendation systems reflects both positive-oriented and concern-related perspectives regarding technological innovation, platform influence, and social implications. This study contributes to a broader understanding of how AI is socially interpreted and represented within contemporary digital platform discourse.</p>
	]]></content:encoded>

	<dc:title>Comparative Keyword Network Analysis of Korean-Language Algorithmic Recommendation Discourses in AI Related to TikTok and YouTube</dc:title>
			<dc:creator>Dae Wan Kim</dc:creator>
			<dc:creator>Luman Dong</dc:creator>
			<dc:creator>Guihua Zhang</dc:creator>
			<dc:creator>Xu Yin</dc:creator>
			<dc:creator>Yujong Hwang</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070238</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-07-16</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-07-16</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>238</prism:startingPage>
		<prism:doi>10.3390/bdcc10070238</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/238</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/237">

	<title>BDCC, Vol. 10, Pages 237: Audio and Video Scene Classification with Cross-Modal Attention Mechanism</title>
	<link>https://www.mdpi.com/2504-2289/10/7/237</link>
	<description>Scene classification aims to identify scene categories by analyzing environmental information. In real-world scenes, scene features are primarily captured through acoustic and visual modalities. However, environmental complexity and information diversity pose significant challenges to classification performance. To address the issues of insufficient information interaction and inadequate feature fusion in traditional segmented fusion methods for audio&amp;amp;ndash;visual data, this paper proposes an audio&amp;amp;ndash;visual scene classification method based on a cross-modal attention mechanism. The fusion mechanism of multi-modal features is investigated, and scene classification performance is enhanced by optimizing the feature fusion strategy. The proposed method consists of three components: a cross-modal attention module, a gating unit, and a residual connection. The cross-modal attention module achieves adaptive feature alignment by establishing dynamic correlations between audio and visual features. The multi-modal gating unit employs an adaptive gating mechanism to dynamically adjust the contribution weight of each modality, thereby alleviating the information loss problem commonly encountered in traditional methods. The residual connection module preserves the original modality features to prevent information degradation. The model&amp;amp;rsquo;s performance is evaluated through testing and validation on real-scene audio&amp;amp;ndash;visual datasets. Multiple sets of experimental results on these datasets demonstrate that the proposed cross-modal attention method achieves a significant improvement in classification accuracy.</description>
	<pubDate>2026-07-15</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 237: Audio and Video Scene Classification with Cross-Modal Attention Mechanism</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/237">doi: 10.3390/bdcc10070237</a></p>
	<p>Authors:
		Mingze Xia
		Guisheng Yin
		Yuxin Dong
		</p>
	<p>Scene classification aims to identify scene categories by analyzing environmental information. In real-world scenes, scene features are primarily captured through acoustic and visual modalities. However, environmental complexity and information diversity pose significant challenges to classification performance. To address the issues of insufficient information interaction and inadequate feature fusion in traditional segmented fusion methods for audio&amp;amp;ndash;visual data, this paper proposes an audio&amp;amp;ndash;visual scene classification method based on a cross-modal attention mechanism. The fusion mechanism of multi-modal features is investigated, and scene classification performance is enhanced by optimizing the feature fusion strategy. The proposed method consists of three components: a cross-modal attention module, a gating unit, and a residual connection. The cross-modal attention module achieves adaptive feature alignment by establishing dynamic correlations between audio and visual features. The multi-modal gating unit employs an adaptive gating mechanism to dynamically adjust the contribution weight of each modality, thereby alleviating the information loss problem commonly encountered in traditional methods. The residual connection module preserves the original modality features to prevent information degradation. The model&amp;amp;rsquo;s performance is evaluated through testing and validation on real-scene audio&amp;amp;ndash;visual datasets. Multiple sets of experimental results on these datasets demonstrate that the proposed cross-modal attention method achieves a significant improvement in classification accuracy.</p>
	]]></content:encoded>

	<dc:title>Audio and Video Scene Classification with Cross-Modal Attention Mechanism</dc:title>
			<dc:creator>Mingze Xia</dc:creator>
			<dc:creator>Guisheng Yin</dc:creator>
			<dc:creator>Yuxin Dong</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070237</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-07-15</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-07-15</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>237</prism:startingPage>
		<prism:doi>10.3390/bdcc10070237</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/237</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/236">

	<title>BDCC, Vol. 10, Pages 236: Big Data- and AI-Driven Hybrid Self-Attention Credit Scoring with Explainable Decisioning</title>
	<link>https://www.mdpi.com/2504-2289/10/7/236</link>
	<description>Real-time retail credit scoring is a data-intensive cognitive computing task. Each decision must fuse heterogeneous signals, execute a non-linear model, return a calibrated probability of default (PD), and emit a regulator-compliant local explanation within milliseconds. We address the most demanding segment of unsecured lending in Kazakhstan&amp;amp;mdash;Salary-Project-Independent (SPI) borrowers, whose principal income stream is not observable by the lender&amp;amp;mdash;and frame scoring as a constrained optimisation problem where we maximise discrimination subject to interpretability, latency, and calibration constraints. We propose a tenure-stratified hybrid framework that couples (i) an online weight-of-evidence logistic regression (WOE-LR) scorecard with (ii) an offline self-attention stacked ensemble (LightGBM, CatBoost, and a tabular self-attention network) whose calibrated PD is quantile-binned, WOE-encoded, and re-injected into the online scorecard as a single auditable predictor. On 551,962 production contracts that originated in 2022&amp;amp;ndash;2024, the repeat-client hybrid attains an area under the receiver operating characteristic curve (AUROC) of 0.826, a Gini coefficient of 0.65, and a Kolmogorov&amp;amp;ndash;Smirnov (KS) statistic of 0.495, preserving roughly half of the offline ensemble&amp;amp;rsquo;s lift over the linear baseline (AUROC 0.79&amp;amp;rarr;0.897) while retaining a fully auditable twelve-coefficient scorecard in production. The new-client scorecard attains an AUROC of 0.741. Non-parametric isotonic recalibration reduces the expected calibration error from 0.27 to below 0.01 and raises the Hosmer&amp;amp;ndash;Lemeshow p-value above 0.99 without altering discrimination. The framework complies with the model risk standards of the Agency of the Republic of Kazakhstan for Regulation and Development of the Financial Market and is delivered as a Spark/MLOps reference architecture, illustrating how big data engineering, attention-based representation learning, and post hoc explanations can be co-designed for a high-stakes, high-throughput, regulated AI application.</description>
	<pubDate>2026-07-13</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 236: Big Data- and AI-Driven Hybrid Self-Attention Credit Scoring with Explainable Decisioning</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/236">doi: 10.3390/bdcc10070236</a></p>
	<p>Authors:
		Gulnaz Zakariya
		Aiman Moldagulova
		Nor’ashikin Ali
		</p>
	<p>Real-time retail credit scoring is a data-intensive cognitive computing task. Each decision must fuse heterogeneous signals, execute a non-linear model, return a calibrated probability of default (PD), and emit a regulator-compliant local explanation within milliseconds. We address the most demanding segment of unsecured lending in Kazakhstan&amp;amp;mdash;Salary-Project-Independent (SPI) borrowers, whose principal income stream is not observable by the lender&amp;amp;mdash;and frame scoring as a constrained optimisation problem where we maximise discrimination subject to interpretability, latency, and calibration constraints. We propose a tenure-stratified hybrid framework that couples (i) an online weight-of-evidence logistic regression (WOE-LR) scorecard with (ii) an offline self-attention stacked ensemble (LightGBM, CatBoost, and a tabular self-attention network) whose calibrated PD is quantile-binned, WOE-encoded, and re-injected into the online scorecard as a single auditable predictor. On 551,962 production contracts that originated in 2022&amp;amp;ndash;2024, the repeat-client hybrid attains an area under the receiver operating characteristic curve (AUROC) of 0.826, a Gini coefficient of 0.65, and a Kolmogorov&amp;amp;ndash;Smirnov (KS) statistic of 0.495, preserving roughly half of the offline ensemble&amp;amp;rsquo;s lift over the linear baseline (AUROC 0.79&amp;amp;rarr;0.897) while retaining a fully auditable twelve-coefficient scorecard in production. The new-client scorecard attains an AUROC of 0.741. Non-parametric isotonic recalibration reduces the expected calibration error from 0.27 to below 0.01 and raises the Hosmer&amp;amp;ndash;Lemeshow p-value above 0.99 without altering discrimination. The framework complies with the model risk standards of the Agency of the Republic of Kazakhstan for Regulation and Development of the Financial Market and is delivered as a Spark/MLOps reference architecture, illustrating how big data engineering, attention-based representation learning, and post hoc explanations can be co-designed for a high-stakes, high-throughput, regulated AI application.</p>
	]]></content:encoded>

	<dc:title>Big Data- and AI-Driven Hybrid Self-Attention Credit Scoring with Explainable Decisioning</dc:title>
			<dc:creator>Gulnaz Zakariya</dc:creator>
			<dc:creator>Aiman Moldagulova</dc:creator>
			<dc:creator>Nor’ashikin Ali</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070236</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-07-13</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-07-13</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>236</prism:startingPage>
		<prism:doi>10.3390/bdcc10070236</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/236</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/235">

	<title>BDCC, Vol. 10, Pages 235: From Static Pages to Symbiotic Intelligence: The CET-WAIP Framework for Understanding AI&amp;ndash;Web Coevolution</title>
	<link>https://www.mdpi.com/2504-2289/10/7/235</link>
	<description>The convergence of the World Wide Web and artificial intelligence (AI) has fundamentally reconfigured digital ecosystems. Yet conventional generational models&amp;amp;mdash;Web 1.0 through 4.0&amp;amp;mdash;remain inadequate for capturing the recursive, mutually constitutive dynamics that characterize this coevolution. The historical trajectory of the Web reveals a progressive expansion of human agency: Web 1.0 afforded read access; Web 2.0 enabled user-generated content; Web 3.0 introduced digital ownership. Artificial intelligence has followed a parallel arc. Early generative systems, such as ChatGPT, demonstrated read capabilities&amp;amp;mdash;conditional upon human authorization. Subsequent code-generation agents, including Codex and Claude Code, extended this to write capabilities&amp;amp;mdash;also contingent upon human approval, initiation, and financial compensation. Web 4.0, however, constitutes a qualitative rupture: AI agents now read, write, own, earn, and transact autonomously, without requiring human oversight. Such automatons operate on their own behalf or on the behalf of a creator who may be human, another agent, or entirely absent. In the Web 4.0 paradigm, the end user is no longer human&amp;amp;mdash;it is AI itself. This review addresses the analytical inadequacy of existing models by introducing the CET-WAIP framework (CoEvolutionary Tiers of Web and AI Paradigms), a novel classificatory framework that models AI&amp;amp;ndash;Web coevolution across seven intelligence tiers spanning infrastructural complexity, cognitive capabilities, and governance dimensions. Grounded in an integrative literature review with PRISMA-informed reporting, the framework aligns key AI paradigms&amp;amp;mdash;from rule-based systems to agentic AI&amp;amp;mdash;with corresponding transformations in Web architecture, revealing how intelligence scaling reshapes user agency, data structures, and ethical oversight. To demonstrate its analytical utility, we conduct a multi-tiered analysis of ChatGPT and compare it with open-source agentic systems (AutoGPT, LangChain, Sora), showing how architectural dissonance between cognitive capabilities and infrastructure is systematically diagnosable. The findings highlight the limitations of linear Web evolution frameworks and underscore the need for intelligence-centric approaches that integrate technological, cognitive, and governance dimensions. We conclude by outlining a research agenda for hybrid intelligence, adaptive governance, and equitable human&amp;amp;ndash;AI collaboration in ecosystems where both human and non-human agents participate as first-class actors.</description>
	<pubDate>2026-07-12</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 235: From Static Pages to Symbiotic Intelligence: The CET-WAIP Framework for Understanding AI&amp;ndash;Web Coevolution</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/235">doi: 10.3390/bdcc10070235</a></p>
	<p>Authors:
		Mohamad Abou Ali
		Fadi Dornaika
		</p>
	<p>The convergence of the World Wide Web and artificial intelligence (AI) has fundamentally reconfigured digital ecosystems. Yet conventional generational models&amp;amp;mdash;Web 1.0 through 4.0&amp;amp;mdash;remain inadequate for capturing the recursive, mutually constitutive dynamics that characterize this coevolution. The historical trajectory of the Web reveals a progressive expansion of human agency: Web 1.0 afforded read access; Web 2.0 enabled user-generated content; Web 3.0 introduced digital ownership. Artificial intelligence has followed a parallel arc. Early generative systems, such as ChatGPT, demonstrated read capabilities&amp;amp;mdash;conditional upon human authorization. Subsequent code-generation agents, including Codex and Claude Code, extended this to write capabilities&amp;amp;mdash;also contingent upon human approval, initiation, and financial compensation. Web 4.0, however, constitutes a qualitative rupture: AI agents now read, write, own, earn, and transact autonomously, without requiring human oversight. Such automatons operate on their own behalf or on the behalf of a creator who may be human, another agent, or entirely absent. In the Web 4.0 paradigm, the end user is no longer human&amp;amp;mdash;it is AI itself. This review addresses the analytical inadequacy of existing models by introducing the CET-WAIP framework (CoEvolutionary Tiers of Web and AI Paradigms), a novel classificatory framework that models AI&amp;amp;ndash;Web coevolution across seven intelligence tiers spanning infrastructural complexity, cognitive capabilities, and governance dimensions. Grounded in an integrative literature review with PRISMA-informed reporting, the framework aligns key AI paradigms&amp;amp;mdash;from rule-based systems to agentic AI&amp;amp;mdash;with corresponding transformations in Web architecture, revealing how intelligence scaling reshapes user agency, data structures, and ethical oversight. To demonstrate its analytical utility, we conduct a multi-tiered analysis of ChatGPT and compare it with open-source agentic systems (AutoGPT, LangChain, Sora), showing how architectural dissonance between cognitive capabilities and infrastructure is systematically diagnosable. The findings highlight the limitations of linear Web evolution frameworks and underscore the need for intelligence-centric approaches that integrate technological, cognitive, and governance dimensions. We conclude by outlining a research agenda for hybrid intelligence, adaptive governance, and equitable human&amp;amp;ndash;AI collaboration in ecosystems where both human and non-human agents participate as first-class actors.</p>
	]]></content:encoded>

	<dc:title>From Static Pages to Symbiotic Intelligence: The CET-WAIP Framework for Understanding AI&amp;amp;ndash;Web Coevolution</dc:title>
			<dc:creator>Mohamad Abou Ali</dc:creator>
			<dc:creator>Fadi Dornaika</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070235</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-07-12</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-07-12</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Review</prism:section>
	<prism:startingPage>235</prism:startingPage>
		<prism:doi>10.3390/bdcc10070235</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/235</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/234">

	<title>BDCC, Vol. 10, Pages 234: Road Inspection 4.0: A Short-Video Benchmark for Deep Learning-Based High-Resolution Pothole Detection in Autonomous Driving</title>
	<link>https://www.mdpi.com/2504-2289/10/7/234</link>
	<description>This work offers an extensive performance evaluation of video-based pothole detection algorithms utilizing a unique dataset of 619 high-resolution movies recorded in South Kalimantan, Indonesia. Seven distinct models were assessed: three multi-frame-based methodologies (Best Frame Selection, Temporal Consistency Loss, and Multi-Frame Ensemble) employing U-Net architectures with temporal modeling, three per-frame models (OneFormer, YOLOv8-seg, and YOLACT), and one fusion ensemble integrating the per-frame models via weighted boxes fusion. The video collection consists of 2 s segments containing 48 frames each, accompanied by ground truth segmentation masks for pothole identification. Results indicate that per-frame models substantially surpass video-based methods, with the fusion ensemble attaining 81% IoU, followed by YOLOv8-seg and OneFormer, each getting 80% IoU. Parameter efficiency investigation indicates that YOLOv8-seg is the most efficient, achieving IoU per million parameters.</description>
	<pubDate>2026-07-10</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 234: Road Inspection 4.0: A Short-Video Benchmark for Deep Learning-Based High-Resolution Pothole Detection in Autonomous Driving</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/234">doi: 10.3390/bdcc10070234</a></p>
	<p>Authors:
		Mohammad Shahin
		Mazdak Maghanaki
		F. Frank Chen
		</p>
	<p>This work offers an extensive performance evaluation of video-based pothole detection algorithms utilizing a unique dataset of 619 high-resolution movies recorded in South Kalimantan, Indonesia. Seven distinct models were assessed: three multi-frame-based methodologies (Best Frame Selection, Temporal Consistency Loss, and Multi-Frame Ensemble) employing U-Net architectures with temporal modeling, three per-frame models (OneFormer, YOLOv8-seg, and YOLACT), and one fusion ensemble integrating the per-frame models via weighted boxes fusion. The video collection consists of 2 s segments containing 48 frames each, accompanied by ground truth segmentation masks for pothole identification. Results indicate that per-frame models substantially surpass video-based methods, with the fusion ensemble attaining 81% IoU, followed by YOLOv8-seg and OneFormer, each getting 80% IoU. Parameter efficiency investigation indicates that YOLOv8-seg is the most efficient, achieving IoU per million parameters.</p>
	]]></content:encoded>

	<dc:title>Road Inspection 4.0: A Short-Video Benchmark for Deep Learning-Based High-Resolution Pothole Detection in Autonomous Driving</dc:title>
			<dc:creator>Mohammad Shahin</dc:creator>
			<dc:creator>Mazdak Maghanaki</dc:creator>
			<dc:creator>F. Frank Chen</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070234</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-07-10</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-07-10</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>234</prism:startingPage>
		<prism:doi>10.3390/bdcc10070234</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/234</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/233">

	<title>BDCC, Vol. 10, Pages 233: AI-Driven Software Testing: A Review</title>
	<link>https://www.mdpi.com/2504-2289/10/7/233</link>
	<description>The rapid evolution of software complexity demands more efficient and autonomous testing mechanisms. Artificial intelligence (AI) has emerged as a solution to the limitations of traditional manual testing in software development, which is time-consuming, prone to human error, and unable to scale with the increasing size and complexity of modern software systems. In this context, this paper presents an application-focused review of 35 selected empirical studies focusing on the use of AI during software testing, based on PRISMA guidelines. We introduce a comprehensive taxonomy categorizing current research into six core fields, including test case generation, defect prediction, and AI model verification. The analysis reveals that large language models, machine learning, and computer vision can significantly improve testing efficiency. Key findings demonstrate that AI can autonomously repair broken test scripts, generate robust synthetic data, enable codeless web testing, and accurately predict system defects before execution. Furthermore, advanced techniques such as reinforcement learning and deep learning successfully validate complex environments, including cloud robotics and quantum software. However, our qualitative and quantitative synthesis also highlights that challenges, such as generative AI &amp;amp;ldquo;hallucinations&amp;amp;rdquo; and the brittleness of Continuous Integration and Continuous Deployment (CI/CD) integration, persist. Ultimately, this review proposes a tailored research roadmap for robust industrial adoption, showing that AI is changing the way software is tested, shifting it from a predominantly reactive and static activity toward a proactive, intelligence-driven discipline.</description>
	<pubDate>2026-07-10</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 233: AI-Driven Software Testing: A Review</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/233">doi: 10.3390/bdcc10070233</a></p>
	<p>Authors:
		Guilherme Martins
		Nelson Tenório
		Jorge Bernardino
		</p>
	<p>The rapid evolution of software complexity demands more efficient and autonomous testing mechanisms. Artificial intelligence (AI) has emerged as a solution to the limitations of traditional manual testing in software development, which is time-consuming, prone to human error, and unable to scale with the increasing size and complexity of modern software systems. In this context, this paper presents an application-focused review of 35 selected empirical studies focusing on the use of AI during software testing, based on PRISMA guidelines. We introduce a comprehensive taxonomy categorizing current research into six core fields, including test case generation, defect prediction, and AI model verification. The analysis reveals that large language models, machine learning, and computer vision can significantly improve testing efficiency. Key findings demonstrate that AI can autonomously repair broken test scripts, generate robust synthetic data, enable codeless web testing, and accurately predict system defects before execution. Furthermore, advanced techniques such as reinforcement learning and deep learning successfully validate complex environments, including cloud robotics and quantum software. However, our qualitative and quantitative synthesis also highlights that challenges, such as generative AI &amp;amp;ldquo;hallucinations&amp;amp;rdquo; and the brittleness of Continuous Integration and Continuous Deployment (CI/CD) integration, persist. Ultimately, this review proposes a tailored research roadmap for robust industrial adoption, showing that AI is changing the way software is tested, shifting it from a predominantly reactive and static activity toward a proactive, intelligence-driven discipline.</p>
	]]></content:encoded>

	<dc:title>AI-Driven Software Testing: A Review</dc:title>
			<dc:creator>Guilherme Martins</dc:creator>
			<dc:creator>Nelson Tenório</dc:creator>
			<dc:creator>Jorge Bernardino</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070233</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-07-10</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-07-10</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Review</prism:section>
	<prism:startingPage>233</prism:startingPage>
		<prism:doi>10.3390/bdcc10070233</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/233</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/232">

	<title>BDCC, Vol. 10, Pages 232: Set Prediction for Outpatient Diagnosis Coding with Sparse Mahalanobis Conformal Scoring</title>
	<link>https://www.mdpi.com/2504-2289/10/7/232</link>
	<description>Diagnosis coding is a large-scale multi-label task in which each clinical encounter may require one or more coding labels from a large label space. Conventional top-k and threshold-based classifiers provide practical coding suggestions but do not directly characterize uncertainty over alternative coding sets. This study proposes sparse Mahalanobis conformal scoring for set prediction in diagnosis coding under extreme multi-label classification (XMC), intended for coding-assist workflows that require compact and reviewable coding suggestions. A sparse XMC model first generates candidate coding labels for each encounter. Candidate label sets are then constructed from the sparse proposal space and scored using a diagonal Mahalanobis nonconformity function calibrated on held-out data. Empirical conformal p-values are assigned to candidate sets, and downstream decision rules are used to obtain a final coding output from the retained region. The framework was evaluated using outpatient EHR data from a tertiary-care hospital, comprising approximately 8.0 million visits from 2018 to 2025 and up to 12,829 diagnosis labels. The primary SMaCS output achieved Micro-F1 close to the strongest threshold-based comparator and the highest exact match ratio among flexible-size decision rules. Compared with the other nonconformity scores, the Mahalanobis score produced a smaller retained region with fewer distinct labels, while preserving the same point-prediction performance. Additional analyses examined conformal region validity, robustness to label-frequency thresholds, code-depth performance, label-frequency subgroups, sample cardinality, department-level variation, and confidence&amp;amp;ndash;credibility stratification. Our results suggest that sparse Mahalanobis conformal scoring provides a useful framework for uncertainty-informed outpatient coding set prediction, while also highlighting the importance of candidate-space adequacy in extreme multi-label diagnosis coding.</description>
	<pubDate>2026-07-10</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 232: Set Prediction for Outpatient Diagnosis Coding with Sparse Mahalanobis Conformal Scoring</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/232">doi: 10.3390/bdcc10070232</a></p>
	<p>Authors:
		Kamonrat Tangudomkit
		Sawrawit Chairat
		Sitthichok Chaichulee
		</p>
	<p>Diagnosis coding is a large-scale multi-label task in which each clinical encounter may require one or more coding labels from a large label space. Conventional top-k and threshold-based classifiers provide practical coding suggestions but do not directly characterize uncertainty over alternative coding sets. This study proposes sparse Mahalanobis conformal scoring for set prediction in diagnosis coding under extreme multi-label classification (XMC), intended for coding-assist workflows that require compact and reviewable coding suggestions. A sparse XMC model first generates candidate coding labels for each encounter. Candidate label sets are then constructed from the sparse proposal space and scored using a diagonal Mahalanobis nonconformity function calibrated on held-out data. Empirical conformal p-values are assigned to candidate sets, and downstream decision rules are used to obtain a final coding output from the retained region. The framework was evaluated using outpatient EHR data from a tertiary-care hospital, comprising approximately 8.0 million visits from 2018 to 2025 and up to 12,829 diagnosis labels. The primary SMaCS output achieved Micro-F1 close to the strongest threshold-based comparator and the highest exact match ratio among flexible-size decision rules. Compared with the other nonconformity scores, the Mahalanobis score produced a smaller retained region with fewer distinct labels, while preserving the same point-prediction performance. Additional analyses examined conformal region validity, robustness to label-frequency thresholds, code-depth performance, label-frequency subgroups, sample cardinality, department-level variation, and confidence&amp;amp;ndash;credibility stratification. Our results suggest that sparse Mahalanobis conformal scoring provides a useful framework for uncertainty-informed outpatient coding set prediction, while also highlighting the importance of candidate-space adequacy in extreme multi-label diagnosis coding.</p>
	]]></content:encoded>

	<dc:title>Set Prediction for Outpatient Diagnosis Coding with Sparse Mahalanobis Conformal Scoring</dc:title>
			<dc:creator>Kamonrat Tangudomkit</dc:creator>
			<dc:creator>Sawrawit Chairat</dc:creator>
			<dc:creator>Sitthichok Chaichulee</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070232</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-07-10</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-07-10</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>232</prism:startingPage>
		<prism:doi>10.3390/bdcc10070232</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/232</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/231">

	<title>BDCC, Vol. 10, Pages 231: DST-Mamba: Spatial&amp;ndash;Temporal Adapter with Boundary-Aware Mamba for Video Temporal Action Detection</title>
	<link>https://www.mdpi.com/2504-2289/10/7/231</link>
	<description>Temporal Action Detection (TAD) localizes and classifies action instances within long videos and underlies many downstream video understanding applications. Transformer-based detectors scale poorly to long sequences due to quadratic self-attention, while existing SSM-based variants tend to dilute fine-grained boundary cues during global modeling. To address these limitations, we propose DST-Mamba, a Decoupled Spatial&amp;amp;ndash;Temporal Mamba Adapter inserted into a frozen video backbone for parameter-efficient end-to-end TAD. DST-Mamba decouples spatial and temporal modeling into two cooperating branches and explicitly fuses them through cross-branch interaction. Within the temporal branch, we introduce a Temporal Boundary-aware SSM (TB-SSM) with direction-specific forward/backward state-transition matrices, providing a stronger inductive bias for asymmetric action boundaries. Across multiple benchmarks, DST-Mamba consistently outperforms competitive Transformer- and SSM-based baselines while being more computationally efficient.</description>
	<pubDate>2026-07-09</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 231: DST-Mamba: Spatial&amp;ndash;Temporal Adapter with Boundary-Aware Mamba for Video Temporal Action Detection</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/231">doi: 10.3390/bdcc10070231</a></p>
	<p>Authors:
		Yicheng Qiu
		Keiji Yanai
		</p>
	<p>Temporal Action Detection (TAD) localizes and classifies action instances within long videos and underlies many downstream video understanding applications. Transformer-based detectors scale poorly to long sequences due to quadratic self-attention, while existing SSM-based variants tend to dilute fine-grained boundary cues during global modeling. To address these limitations, we propose DST-Mamba, a Decoupled Spatial&amp;amp;ndash;Temporal Mamba Adapter inserted into a frozen video backbone for parameter-efficient end-to-end TAD. DST-Mamba decouples spatial and temporal modeling into two cooperating branches and explicitly fuses them through cross-branch interaction. Within the temporal branch, we introduce a Temporal Boundary-aware SSM (TB-SSM) with direction-specific forward/backward state-transition matrices, providing a stronger inductive bias for asymmetric action boundaries. Across multiple benchmarks, DST-Mamba consistently outperforms competitive Transformer- and SSM-based baselines while being more computationally efficient.</p>
	]]></content:encoded>

	<dc:title>DST-Mamba: Spatial&amp;amp;ndash;Temporal Adapter with Boundary-Aware Mamba for Video Temporal Action Detection</dc:title>
			<dc:creator>Yicheng Qiu</dc:creator>
			<dc:creator>Keiji Yanai</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070231</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-07-09</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-07-09</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>231</prism:startingPage>
		<prism:doi>10.3390/bdcc10070231</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/231</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/230">

	<title>BDCC, Vol. 10, Pages 230: HSIC-DIMFMC: A Multi-View Functional Matrix Completion Method with Dual-Information Graph Regularization for Meteorological Data Imputation</title>
	<link>https://www.mdpi.com/2504-2289/10/7/230</link>
	<description>Continuous and complete meteorological observations are essential for reliable climate analysis and environmental assessment. However, missing values caused by sensor malfunctions and transmission failures can introduce systematic biases and increase uncertainty in downstream applications. Meteorological variables can be modeled as functional data and typically exhibit nonlinear inter-variable dependencies alongside temporal smoothness; these properties provide valuable prior information for missing data recovery. To address this issue, we propose HSIC-DIMFMC, a multi-view functional matrix completion method for meteorological data imputation that integrates the Hilbert&amp;amp;ndash;Schmidt Independence Criterion (HSIC) and dual-information graph regularization. Within a unified framework of functional data analysis and multi-view learning, HSIC is utilized to capture nonlinear dependencies across multiple views, while dual-information graph regularization preserves local structural relationships and temporal smoothness. This joint modeling strategy significantly improves latent representation learning and enhances imputation performance. Experiments on real meteorological datasets demonstrate that the proposed method consistently outperforms several state-of-the-art baselines, especially for strongly correlated variable pairs such as temperature&amp;amp;ndash;dew point and wind speed&amp;amp;ndash;maximum wind speed. Compared with seven representative approaches&amp;amp;mdash;ranging from traditional spatial interpolation to advanced functional matrix completion models&amp;amp;mdash;HSIC-DIMFMC achieves average reductions of 43.97&amp;amp;ndash;73.59% in RMSE. The results indicate that HSIC-DIMFMC effectively exploits nonlinear cross-view dependencies and structural information, providing a robust solution for collaborative meteorological data imputation.</description>
	<pubDate>2026-07-08</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 230: HSIC-DIMFMC: A Multi-View Functional Matrix Completion Method with Dual-Information Graph Regularization for Meteorological Data Imputation</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/230">doi: 10.3390/bdcc10070230</a></p>
	<p>Authors:
		Haiyan Gao
		Youdi Bian
		</p>
	<p>Continuous and complete meteorological observations are essential for reliable climate analysis and environmental assessment. However, missing values caused by sensor malfunctions and transmission failures can introduce systematic biases and increase uncertainty in downstream applications. Meteorological variables can be modeled as functional data and typically exhibit nonlinear inter-variable dependencies alongside temporal smoothness; these properties provide valuable prior information for missing data recovery. To address this issue, we propose HSIC-DIMFMC, a multi-view functional matrix completion method for meteorological data imputation that integrates the Hilbert&amp;amp;ndash;Schmidt Independence Criterion (HSIC) and dual-information graph regularization. Within a unified framework of functional data analysis and multi-view learning, HSIC is utilized to capture nonlinear dependencies across multiple views, while dual-information graph regularization preserves local structural relationships and temporal smoothness. This joint modeling strategy significantly improves latent representation learning and enhances imputation performance. Experiments on real meteorological datasets demonstrate that the proposed method consistently outperforms several state-of-the-art baselines, especially for strongly correlated variable pairs such as temperature&amp;amp;ndash;dew point and wind speed&amp;amp;ndash;maximum wind speed. Compared with seven representative approaches&amp;amp;mdash;ranging from traditional spatial interpolation to advanced functional matrix completion models&amp;amp;mdash;HSIC-DIMFMC achieves average reductions of 43.97&amp;amp;ndash;73.59% in RMSE. The results indicate that HSIC-DIMFMC effectively exploits nonlinear cross-view dependencies and structural information, providing a robust solution for collaborative meteorological data imputation.</p>
	]]></content:encoded>

	<dc:title>HSIC-DIMFMC: A Multi-View Functional Matrix Completion Method with Dual-Information Graph Regularization for Meteorological Data Imputation</dc:title>
			<dc:creator>Haiyan Gao</dc:creator>
			<dc:creator>Youdi Bian</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070230</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-07-08</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-07-08</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>230</prism:startingPage>
		<prism:doi>10.3390/bdcc10070230</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/230</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/229">

	<title>BDCC, Vol. 10, Pages 229: Anticipatory AI Governance in the Age of Supercomputing: A Mixed-Methods Multistakeholder Approach in the Basque Country</title>
	<link>https://www.mdpi.com/2504-2289/10/7/229</link>
	<description>Artificial Intelligence (AI) is increasingly embedded in public governance, raising new challenges for anticipating its societal implications while safeguarding democratic accountability within expanding computational infrastructures. This article examines how anticipatory AI governance can be operationalised in the age of supercomputing through a mixed-methods, multistakeholder study conducted in the Basque Country (Spain). The empirical focus is Gipuzkoa, a devolved historical territory with fiscal autonomy and a rapidly developing advanced-computing ecosystem centred in Donostia&amp;amp;ndash;San Sebasti&amp;amp;aacute;n, where regional initiatives are positioning the territory within Europe&amp;amp;rsquo;s emerging high-performance and quantum computing landscape. The study combines participatory action research involving six civil society organisations, seven provincial directorates, and eleven municipalities with an online citizen survey (N = 911). The findings indicate that anticipatory AI governance is supported through four interrelated governance mechanisms: institutional coordination across administrative levels, multistakeholder participation, territorial public capability, and the strategic embedding of advanced computational infrastructures. Rather than evaluating the governance of supercomputing technologies themselves, the analysis examines governance perceptions, institutional practices, and democratic arrangements associated with these infrastructures. The article&amp;amp;rsquo;s contribution lies in integrating anticipatory AI governance, territorial governance, and advanced computational infrastructures within a devolved city-regional setting, offering evidence-informed insights for regions seeking to strengthen democratic capacity alongside technological innovation.</description>
	<pubDate>2026-07-08</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 229: Anticipatory AI Governance in the Age of Supercomputing: A Mixed-Methods Multistakeholder Approach in the Basque Country</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/229">doi: 10.3390/bdcc10070229</a></p>
	<p>Authors:
		Igor Calzada
		Itziar Eizaguirre
		</p>
	<p>Artificial Intelligence (AI) is increasingly embedded in public governance, raising new challenges for anticipating its societal implications while safeguarding democratic accountability within expanding computational infrastructures. This article examines how anticipatory AI governance can be operationalised in the age of supercomputing through a mixed-methods, multistakeholder study conducted in the Basque Country (Spain). The empirical focus is Gipuzkoa, a devolved historical territory with fiscal autonomy and a rapidly developing advanced-computing ecosystem centred in Donostia&amp;amp;ndash;San Sebasti&amp;amp;aacute;n, where regional initiatives are positioning the territory within Europe&amp;amp;rsquo;s emerging high-performance and quantum computing landscape. The study combines participatory action research involving six civil society organisations, seven provincial directorates, and eleven municipalities with an online citizen survey (N = 911). The findings indicate that anticipatory AI governance is supported through four interrelated governance mechanisms: institutional coordination across administrative levels, multistakeholder participation, territorial public capability, and the strategic embedding of advanced computational infrastructures. Rather than evaluating the governance of supercomputing technologies themselves, the analysis examines governance perceptions, institutional practices, and democratic arrangements associated with these infrastructures. The article&amp;amp;rsquo;s contribution lies in integrating anticipatory AI governance, territorial governance, and advanced computational infrastructures within a devolved city-regional setting, offering evidence-informed insights for regions seeking to strengthen democratic capacity alongside technological innovation.</p>
	]]></content:encoded>

	<dc:title>Anticipatory AI Governance in the Age of Supercomputing: A Mixed-Methods Multistakeholder Approach in the Basque Country</dc:title>
			<dc:creator>Igor Calzada</dc:creator>
			<dc:creator>Itziar Eizaguirre</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070229</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-07-08</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-07-08</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>229</prism:startingPage>
		<prism:doi>10.3390/bdcc10070229</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/229</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/228">

	<title>BDCC, Vol. 10, Pages 228: CorrelaCache: A Cache Replacement Model Based on Imitation Learning and Autocorrelation Mechanism</title>
	<link>https://www.mdpi.com/2504-2289/10/7/228</link>
	<description>Existing cache replacement strategies in large-scale spatiotemporal data systems struggle to cope with complex and dynamic access patterns characterized by long-tail distributions and periodic behaviors. Traditional heuristic-based methods such as Least Recently Used (LRU) and Least Frequently Used (LFU) frequently fail to generalize across varying workloads, while recent learning-based approaches are limited by their reliance on hand-crafted features or short-term dependencies. In this paper, we propose a cache replacement framework named CorrelaCache, which integrates imitation learning with a temporal autocorrelation mechanism to capture both short-term and long-range periodic access patterns. By modeling the replacement task as a Markov Decision Process (MDP) and using the Belady optimal policy as the supervision signal, our method adopts Long Short-Term Memory (LSTM) networks for sequential encoding and employs Fast Fourier Transform (FFT)-based autocorrelation to detect and align periodic phases in access history. We further incorporate a joint prediction layer and a hybrid loss function that combines ranking loss and reuse distance prediction loss, and mitigate distributional shift during training via the Dataset Aggregation (DAgger) algorithm. Experimental results on five public meteorological datasets with generated hydrological access traces show that CorrelaCache outperforms representative baselines in the evaluated workloads.</description>
	<pubDate>2026-07-07</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 228: CorrelaCache: A Cache Replacement Model Based on Imitation Learning and Autocorrelation Mechanism</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/228">doi: 10.3390/bdcc10070228</a></p>
	<p>Authors:
		Shuaijie Wu
		Zekun Yan
		Hao Gui
		Ruoshan Kong
		Hua Chen
		Feng Liu
		</p>
	<p>Existing cache replacement strategies in large-scale spatiotemporal data systems struggle to cope with complex and dynamic access patterns characterized by long-tail distributions and periodic behaviors. Traditional heuristic-based methods such as Least Recently Used (LRU) and Least Frequently Used (LFU) frequently fail to generalize across varying workloads, while recent learning-based approaches are limited by their reliance on hand-crafted features or short-term dependencies. In this paper, we propose a cache replacement framework named CorrelaCache, which integrates imitation learning with a temporal autocorrelation mechanism to capture both short-term and long-range periodic access patterns. By modeling the replacement task as a Markov Decision Process (MDP) and using the Belady optimal policy as the supervision signal, our method adopts Long Short-Term Memory (LSTM) networks for sequential encoding and employs Fast Fourier Transform (FFT)-based autocorrelation to detect and align periodic phases in access history. We further incorporate a joint prediction layer and a hybrid loss function that combines ranking loss and reuse distance prediction loss, and mitigate distributional shift during training via the Dataset Aggregation (DAgger) algorithm. Experimental results on five public meteorological datasets with generated hydrological access traces show that CorrelaCache outperforms representative baselines in the evaluated workloads.</p>
	]]></content:encoded>

	<dc:title>CorrelaCache: A Cache Replacement Model Based on Imitation Learning and Autocorrelation Mechanism</dc:title>
			<dc:creator>Shuaijie Wu</dc:creator>
			<dc:creator>Zekun Yan</dc:creator>
			<dc:creator>Hao Gui</dc:creator>
			<dc:creator>Ruoshan Kong</dc:creator>
			<dc:creator>Hua Chen</dc:creator>
			<dc:creator>Feng Liu</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070228</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-07-07</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-07-07</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>228</prism:startingPage>
		<prism:doi>10.3390/bdcc10070228</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/228</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/227">

	<title>BDCC, Vol. 10, Pages 227: A Data-Driven Real-Time Fall-from-Height Detection Method for On-Device Worker Safety Wearables</title>
	<link>https://www.mdpi.com/2504-2289/10/7/227</link>
	<description>Fall-from-height (FFH) detection is a critical component in wearable safety systems, particularly in environments where high-intensity movements can lead to frequent false positives. Conventional approaches based on simple thresholding of acceleration signals often fail to reliably distinguish FFH events from non-fall activities due to overlapping signal characteristics. This paper proposes a data-driven FFH detection method that integrates multiple complementary features into a unified score-based model. The proposed approach first performs structured peak detection to extract candidate impact events while significantly reducing the number of samples requiring further processing. Each candidate is then evaluated using pre-peak structure, post-impact stability, and pressure variation, which respectively capture structural, temporal, and physical characteristics of FFH events. Based on statistical analysis, feature-wise score contributions are designed to reflect their discriminative strength, and the final FFH decision is performed using an additive scoring mechanism. This formulation enables flexible handling of ambiguous cases while preserving strong FFH characteristics. Experimental results demonstrate that the proposed method maintains 100% recall at the selected decision threshold while significantly reducing false positives from non-FFH activities. In addition, the peak detection stage reduces more than 99% of raw samples, enabling efficient on-device processing suitable for wearable systems. The proposed method also includes quantitative analysis of latency characteristics. Although FFH inference latency is influenced by asynchronous pressure sensing, the delay remains bounded and predictable, and most detections are completed within a practical time range for real-time wearable safety applications. Overall, the proposed method achieves a practical balance between detection sensitivity, false-positive suppression, computational efficiency, and real-time feasibility, demonstrating its applicability to wearable safety systems.</description>
	<pubDate>2026-07-06</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 227: A Data-Driven Real-Time Fall-from-Height Detection Method for On-Device Worker Safety Wearables</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/227">doi: 10.3390/bdcc10070227</a></p>
	<p>Authors:
		SangHyeok Kim
		Daejin Park
		Soon Ju Kang
		</p>
	<p>Fall-from-height (FFH) detection is a critical component in wearable safety systems, particularly in environments where high-intensity movements can lead to frequent false positives. Conventional approaches based on simple thresholding of acceleration signals often fail to reliably distinguish FFH events from non-fall activities due to overlapping signal characteristics. This paper proposes a data-driven FFH detection method that integrates multiple complementary features into a unified score-based model. The proposed approach first performs structured peak detection to extract candidate impact events while significantly reducing the number of samples requiring further processing. Each candidate is then evaluated using pre-peak structure, post-impact stability, and pressure variation, which respectively capture structural, temporal, and physical characteristics of FFH events. Based on statistical analysis, feature-wise score contributions are designed to reflect their discriminative strength, and the final FFH decision is performed using an additive scoring mechanism. This formulation enables flexible handling of ambiguous cases while preserving strong FFH characteristics. Experimental results demonstrate that the proposed method maintains 100% recall at the selected decision threshold while significantly reducing false positives from non-FFH activities. In addition, the peak detection stage reduces more than 99% of raw samples, enabling efficient on-device processing suitable for wearable systems. The proposed method also includes quantitative analysis of latency characteristics. Although FFH inference latency is influenced by asynchronous pressure sensing, the delay remains bounded and predictable, and most detections are completed within a practical time range for real-time wearable safety applications. Overall, the proposed method achieves a practical balance between detection sensitivity, false-positive suppression, computational efficiency, and real-time feasibility, demonstrating its applicability to wearable safety systems.</p>
	]]></content:encoded>

	<dc:title>A Data-Driven Real-Time Fall-from-Height Detection Method for On-Device Worker Safety Wearables</dc:title>
			<dc:creator>SangHyeok Kim</dc:creator>
			<dc:creator>Daejin Park</dc:creator>
			<dc:creator>Soon Ju Kang</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070227</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-07-06</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-07-06</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>227</prism:startingPage>
		<prism:doi>10.3390/bdcc10070227</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/227</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/226">

	<title>BDCC, Vol. 10, Pages 226: Snapshot-Based Analysis of Distributed Organizational and Technical System</title>
	<link>https://www.mdpi.com/2504-2289/10/7/226</link>
	<description>Construction companies, petrochemical enterprises, and airports are examples of large-scale organizational&amp;amp;ndash;technical systems (OTSs) and are characterized by a distributed structure, numerous parallel technological and business processes, and substantial energy consumption. The control of such systems is implemented through hierarchical distributed systems that require the regular collection, synchronization, and analysis of large volumes of heterogeneous data. This paper proposes a methodology for performance analysis and energy consumption optimization in OTSs based on the combined use of hierarchical control, business process modeling in BPMN and DRAKON notations, and the use of snapshots&amp;amp;mdash;consistent global states of a distributed system captured at specified time instants. The specifics of snapshot generation algorithms are discussed, including copy-on-write, the Chandy&amp;amp;ndash;Lamport algorithm, cloud orchestration, and log-based point-in-time recovery. A snapshot acquisition optimization problem is formulated, which minimizes the deviation of the captured state from the actual state under constraints on frequency, synchronization delay, and cost. The feasibility of the approach is illustrated by a numerical example of energy redistribution between the levels of a hierarchical control system using distributed model predictive control (DMPC). The advantages of the method include obtaining an objective &amp;amp;ldquo;as is&amp;amp;rdquo; picture, the applicability of control-theoretic methods for distributed systems based on big data processing, the ability to localize faulty subsystems, and its utility in assessing a company&amp;amp;rsquo;s condition for stakeholders.</description>
	<pubDate>2026-07-06</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 226: Snapshot-Based Analysis of Distributed Organizational and Technical System</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/226">doi: 10.3390/bdcc10070226</a></p>
	<p>Authors:
		Sagit Valeev
		Natalya Kondratyeva
		</p>
	<p>Construction companies, petrochemical enterprises, and airports are examples of large-scale organizational&amp;amp;ndash;technical systems (OTSs) and are characterized by a distributed structure, numerous parallel technological and business processes, and substantial energy consumption. The control of such systems is implemented through hierarchical distributed systems that require the regular collection, synchronization, and analysis of large volumes of heterogeneous data. This paper proposes a methodology for performance analysis and energy consumption optimization in OTSs based on the combined use of hierarchical control, business process modeling in BPMN and DRAKON notations, and the use of snapshots&amp;amp;mdash;consistent global states of a distributed system captured at specified time instants. The specifics of snapshot generation algorithms are discussed, including copy-on-write, the Chandy&amp;amp;ndash;Lamport algorithm, cloud orchestration, and log-based point-in-time recovery. A snapshot acquisition optimization problem is formulated, which minimizes the deviation of the captured state from the actual state under constraints on frequency, synchronization delay, and cost. The feasibility of the approach is illustrated by a numerical example of energy redistribution between the levels of a hierarchical control system using distributed model predictive control (DMPC). The advantages of the method include obtaining an objective &amp;amp;ldquo;as is&amp;amp;rdquo; picture, the applicability of control-theoretic methods for distributed systems based on big data processing, the ability to localize faulty subsystems, and its utility in assessing a company&amp;amp;rsquo;s condition for stakeholders.</p>
	]]></content:encoded>

	<dc:title>Snapshot-Based Analysis of Distributed Organizational and Technical System</dc:title>
			<dc:creator>Sagit Valeev</dc:creator>
			<dc:creator>Natalya Kondratyeva</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070226</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-07-06</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-07-06</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>226</prism:startingPage>
		<prism:doi>10.3390/bdcc10070226</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/226</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/225">

	<title>BDCC, Vol. 10, Pages 225: CTA-Net: A Cross-Temporal Attention Network for Change Detection in Remote Sensing Imagery</title>
	<link>https://www.mdpi.com/2504-2289/10/7/225</link>
	<description>Accurate change detection in high-resolution remote sensing imagery is essential for urban planning, land-use monitoring, and disaster response. This study introduces CTA-Net, a Cross-Temporal Attention Network for binary change detection in bi-temporal optical imagery, designed to improve robustness against pseudo-changes caused by illumination variation, seasonal effects, and sensor noise. The proposed method employs a shared Siamese encoder with multi-scale Cross-Temporal Attention modules that derive spatial and channel attention from L2 feature differences, along with a lightweight confidence estimation head for per-pixel uncertainty modelling. A hybrid loss function combining confidence-weighted binary cross-entropy and focal loss is used to address class imbalance. Experiments on the LEVIR-CD dataset demonstrate that CTA-Net achieves an overall accuracy of 98.99%, an F1-score of 87.68%, an Intersection over Union of 78.06%, a Cohen&amp;amp;rsquo;s kappa of 0.8715, and a Matthews Correlation Coefficient of 0.8721, with stable convergence and minimal overfitting. Qualitative and calibration analyses further indicate that the model produces interpretable attention maps and reliable probabilistic outputs. To evaluate cross-domain generalization, we conduct a transfer learning case study on multispectral Sentinel-2 agricultural imagery. The model is adapted to 11-channel input and fine-tuned on automatically generated change masks derived from NDVI-delta thresholding. Under this supervision protocol, CTA-Net achieves an F1-score of 95.18% and an IoU of 90.81% on a held-out test region, with balanced precision and recall. While these results demonstrate effective adaptation across sensor modality, spatial resolution, and semantic domain, the evaluation reflects agreement with the mask generation procedure rather than independently annotated ground truth. While CTA-Net shows strong performance and reasonable interpretability, its cross-domain evaluation is limited by the use of automatically generated labels. As a result, the reported transferability should be interpreted cautiously until validated on human-annotated datasets.</description>
	<pubDate>2026-07-06</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 225: CTA-Net: A Cross-Temporal Attention Network for Change Detection in Remote Sensing Imagery</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/225">doi: 10.3390/bdcc10070225</a></p>
	<p>Authors:
		Azamat Serek
		Farida Abdoldina
		Mukhtarov Asylbek
		Valentin Smurygin
		Gulnaz Nabiyeva
		</p>
	<p>Accurate change detection in high-resolution remote sensing imagery is essential for urban planning, land-use monitoring, and disaster response. This study introduces CTA-Net, a Cross-Temporal Attention Network for binary change detection in bi-temporal optical imagery, designed to improve robustness against pseudo-changes caused by illumination variation, seasonal effects, and sensor noise. The proposed method employs a shared Siamese encoder with multi-scale Cross-Temporal Attention modules that derive spatial and channel attention from L2 feature differences, along with a lightweight confidence estimation head for per-pixel uncertainty modelling. A hybrid loss function combining confidence-weighted binary cross-entropy and focal loss is used to address class imbalance. Experiments on the LEVIR-CD dataset demonstrate that CTA-Net achieves an overall accuracy of 98.99%, an F1-score of 87.68%, an Intersection over Union of 78.06%, a Cohen&amp;amp;rsquo;s kappa of 0.8715, and a Matthews Correlation Coefficient of 0.8721, with stable convergence and minimal overfitting. Qualitative and calibration analyses further indicate that the model produces interpretable attention maps and reliable probabilistic outputs. To evaluate cross-domain generalization, we conduct a transfer learning case study on multispectral Sentinel-2 agricultural imagery. The model is adapted to 11-channel input and fine-tuned on automatically generated change masks derived from NDVI-delta thresholding. Under this supervision protocol, CTA-Net achieves an F1-score of 95.18% and an IoU of 90.81% on a held-out test region, with balanced precision and recall. While these results demonstrate effective adaptation across sensor modality, spatial resolution, and semantic domain, the evaluation reflects agreement with the mask generation procedure rather than independently annotated ground truth. While CTA-Net shows strong performance and reasonable interpretability, its cross-domain evaluation is limited by the use of automatically generated labels. As a result, the reported transferability should be interpreted cautiously until validated on human-annotated datasets.</p>
	]]></content:encoded>

	<dc:title>CTA-Net: A Cross-Temporal Attention Network for Change Detection in Remote Sensing Imagery</dc:title>
			<dc:creator>Azamat Serek</dc:creator>
			<dc:creator>Farida Abdoldina</dc:creator>
			<dc:creator>Mukhtarov Asylbek</dc:creator>
			<dc:creator>Valentin Smurygin</dc:creator>
			<dc:creator>Gulnaz Nabiyeva</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070225</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-07-06</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-07-06</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>225</prism:startingPage>
		<prism:doi>10.3390/bdcc10070225</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/225</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/224">

	<title>BDCC, Vol. 10, Pages 224: Better Prompts, Better Usefulness: A Systematic Review and Experimental Evaluation of Structured Prompting Techniques in Large Language Models</title>
	<link>https://www.mdpi.com/2504-2289/10/7/224</link>
	<description>Large Language Models (LLMs) have rapidly become central components of cognitive computing systems and AI-assisted knowledge work. However, the effectiveness of LLM-generated outputs depends not only on the model&amp;amp;rsquo;s capabilities but also on the structure of the prompts used to guide them. This study investigates how structured prompting techniques influence perceived output usefulness in business-oriented tasks. First, we conduct a systematic literature review following PRISMA guidelines to identify, classify, and synthesize existing prompt enhancement strategies. The review leads to the development of a taxonomy distinguishing task-alignment techniques (e.g., one-shot and few-shot prompting) from reasoning-transparency techniques (e.g., Chain-of-Thought prompting). Building on this taxonomy, we design a controlled experimental study in which knowledge workers evaluate LLM-generated outputs across analytical and summarization tasks. Using linear mixed-effects modeling, we assess the impact of prompting techniques and the moderating role of Generative AI usage frequency. Results indicate that structured prompting significantly increases perceived usefulness compared to baseline approaches, with the combination of example-based conditioning and explicit reasoning scaffolding yielding the highest evaluations. The moderating effect of usage frequency is not statistically significant, suggesting that the benefits of structured prompt design are robust across different experience levels. These findings position prompt structure as a practical cognitive interface mechanism and provide evidence-based guidelines for enhancing human&amp;amp;ndash;AI interaction in cognitive computing environments.</description>
	<pubDate>2026-07-06</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 224: Better Prompts, Better Usefulness: A Systematic Review and Experimental Evaluation of Structured Prompting Techniques in Large Language Models</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/224">doi: 10.3390/bdcc10070224</a></p>
	<p>Authors:
		Alessia Cantini
		Andrea De Mauro
		</p>
	<p>Large Language Models (LLMs) have rapidly become central components of cognitive computing systems and AI-assisted knowledge work. However, the effectiveness of LLM-generated outputs depends not only on the model&amp;amp;rsquo;s capabilities but also on the structure of the prompts used to guide them. This study investigates how structured prompting techniques influence perceived output usefulness in business-oriented tasks. First, we conduct a systematic literature review following PRISMA guidelines to identify, classify, and synthesize existing prompt enhancement strategies. The review leads to the development of a taxonomy distinguishing task-alignment techniques (e.g., one-shot and few-shot prompting) from reasoning-transparency techniques (e.g., Chain-of-Thought prompting). Building on this taxonomy, we design a controlled experimental study in which knowledge workers evaluate LLM-generated outputs across analytical and summarization tasks. Using linear mixed-effects modeling, we assess the impact of prompting techniques and the moderating role of Generative AI usage frequency. Results indicate that structured prompting significantly increases perceived usefulness compared to baseline approaches, with the combination of example-based conditioning and explicit reasoning scaffolding yielding the highest evaluations. The moderating effect of usage frequency is not statistically significant, suggesting that the benefits of structured prompt design are robust across different experience levels. These findings position prompt structure as a practical cognitive interface mechanism and provide evidence-based guidelines for enhancing human&amp;amp;ndash;AI interaction in cognitive computing environments.</p>
	]]></content:encoded>

	<dc:title>Better Prompts, Better Usefulness: A Systematic Review and Experimental Evaluation of Structured Prompting Techniques in Large Language Models</dc:title>
			<dc:creator>Alessia Cantini</dc:creator>
			<dc:creator>Andrea De Mauro</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070224</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-07-06</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-07-06</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Systematic Review</prism:section>
	<prism:startingPage>224</prism:startingPage>
		<prism:doi>10.3390/bdcc10070224</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/224</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/223">

	<title>BDCC, Vol. 10, Pages 223: Seasonal Variation in Heart Rate Variability Associated with Physical Activity and Regional Variability Observed in the ALLSTAR Holter ECG Database</title>
	<link>https://www.mdpi.com/2504-2289/10/7/223</link>
	<description>Seasonal variation in heart rate variability (HRV) reflects multiple interacting determinants rather than a single underlying determinant. In this study, we aimed to examine subgroup-level seasonal HRV variation in relation to physical activity (PA) using large-scale real-world data. We analyzed 133,747 24-h ECG recordings with tri-axial accelerometry from the ALLSTAR database across eight regions in Japan (after excluding regions with insufficient sample sizes) collected between 2015 and 2021. Seasonal variation (&amp;amp;Delta;) was defined as the difference between the maximum and minimum seasonal mean values. Weighted least squares models (WLS) were applied to examine associations between &amp;amp;Delta;PA and multiple HRV indices, including interaction terms for sex and region, while regional differences in residual variability were assessed using Levene&amp;amp;rsquo;s test. During the normal period, significant associations between &amp;amp;Delta;PA and &amp;amp;Delta;HRV were observed for specific indices (&amp;amp;Delta;ULF, &amp;amp;Delta;VLF, &amp;amp;Delta;HF, and &amp;amp;Delta;LF/HF), whereas other indices were not significant. During the Coronavirus Disease 2019 (COVID-19) period, significant associations were observed for &amp;amp;Delta;RRI, &amp;amp;Delta;SDRR, and &amp;amp;Delta;LF/HF, indicating that the association between PA and seasonal HRV variation was index-specific. Sex interactions were not statistically significant after FDR (False Discovery Rate) correction in either period, suggesting a limited role of sex in the PA&amp;amp;ndash;HRV relationship at the population level. Regional differences in HRV sensitivity to PA were statistically significant but heterogeneous across regions. In contrast, residual variability exhibited significant regional differences across multiple HRV indices in both periods. These patterns were not fully explained by sample size and showed stable regional heterogeneity. These findings suggest that subgroup-level regional heterogeneity in seasonal HRV variation is primarily reflected in the unexplained component rather than in the direct PA&amp;amp;ndash;HRV relationship, indicating the presence of region-specific variability in the unexplained component beyond behavioral influences.</description>
	<pubDate>2026-07-06</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 223: Seasonal Variation in Heart Rate Variability Associated with Physical Activity and Regional Variability Observed in the ALLSTAR Holter ECG Database</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/223">doi: 10.3390/bdcc10070223</a></p>
	<p>Authors:
		Yutaka Yoshida
		Junichiro Hayano
		</p>
	<p>Seasonal variation in heart rate variability (HRV) reflects multiple interacting determinants rather than a single underlying determinant. In this study, we aimed to examine subgroup-level seasonal HRV variation in relation to physical activity (PA) using large-scale real-world data. We analyzed 133,747 24-h ECG recordings with tri-axial accelerometry from the ALLSTAR database across eight regions in Japan (after excluding regions with insufficient sample sizes) collected between 2015 and 2021. Seasonal variation (&amp;amp;Delta;) was defined as the difference between the maximum and minimum seasonal mean values. Weighted least squares models (WLS) were applied to examine associations between &amp;amp;Delta;PA and multiple HRV indices, including interaction terms for sex and region, while regional differences in residual variability were assessed using Levene&amp;amp;rsquo;s test. During the normal period, significant associations between &amp;amp;Delta;PA and &amp;amp;Delta;HRV were observed for specific indices (&amp;amp;Delta;ULF, &amp;amp;Delta;VLF, &amp;amp;Delta;HF, and &amp;amp;Delta;LF/HF), whereas other indices were not significant. During the Coronavirus Disease 2019 (COVID-19) period, significant associations were observed for &amp;amp;Delta;RRI, &amp;amp;Delta;SDRR, and &amp;amp;Delta;LF/HF, indicating that the association between PA and seasonal HRV variation was index-specific. Sex interactions were not statistically significant after FDR (False Discovery Rate) correction in either period, suggesting a limited role of sex in the PA&amp;amp;ndash;HRV relationship at the population level. Regional differences in HRV sensitivity to PA were statistically significant but heterogeneous across regions. In contrast, residual variability exhibited significant regional differences across multiple HRV indices in both periods. These patterns were not fully explained by sample size and showed stable regional heterogeneity. These findings suggest that subgroup-level regional heterogeneity in seasonal HRV variation is primarily reflected in the unexplained component rather than in the direct PA&amp;amp;ndash;HRV relationship, indicating the presence of region-specific variability in the unexplained component beyond behavioral influences.</p>
	]]></content:encoded>

	<dc:title>Seasonal Variation in Heart Rate Variability Associated with Physical Activity and Regional Variability Observed in the ALLSTAR Holter ECG Database</dc:title>
			<dc:creator>Yutaka Yoshida</dc:creator>
			<dc:creator>Junichiro Hayano</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070223</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-07-06</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-07-06</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>223</prism:startingPage>
		<prism:doi>10.3390/bdcc10070223</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/223</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/222">

	<title>BDCC, Vol. 10, Pages 222: Economical, Optimal and Uncertain Multiple-View L2 Triangulation via LMIs</title>
	<link>https://www.mdpi.com/2504-2289/10/7/222</link>
	<description>This paper proposes a novel approach for multiple-view L2 triangulation, a key problem in computer vision which consists of estimating a scene point from its estimated image projections on two or more cameras and from the estimated projection matrices of the cameras by minimizing the reprojection error in the L2 norm. In the proposed approach, the estimated image projections are allowed to be uncertain in admissible regions described by polynomial inequalities and equalities, and an estimate of the scene point is obtained by solving a linear matrix inequality (LMI) problem built with matrix decompositions, polynomial multipliers, and the Gram matrix method. It is proven that the optimal estimate can always be achieved by using multipliers with sufficiently large degree. Moreover, a simple test is provided in order to establish the optimality of the obtained estimate. As shown by some examples with real and synthetic data, the proposed approach presents key advantages with respect to several existing methods of a different nature, which may fail to find the optimal estimate, may not allow one to establish the optimality of the found estimate, or may require a larger computational burden.</description>
	<pubDate>2026-07-05</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 222: Economical, Optimal and Uncertain Multiple-View L2 Triangulation via LMIs</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/222">doi: 10.3390/bdcc10070222</a></p>
	<p>Authors:
		Graziano Chesi
		</p>
	<p>This paper proposes a novel approach for multiple-view L2 triangulation, a key problem in computer vision which consists of estimating a scene point from its estimated image projections on two or more cameras and from the estimated projection matrices of the cameras by minimizing the reprojection error in the L2 norm. In the proposed approach, the estimated image projections are allowed to be uncertain in admissible regions described by polynomial inequalities and equalities, and an estimate of the scene point is obtained by solving a linear matrix inequality (LMI) problem built with matrix decompositions, polynomial multipliers, and the Gram matrix method. It is proven that the optimal estimate can always be achieved by using multipliers with sufficiently large degree. Moreover, a simple test is provided in order to establish the optimality of the obtained estimate. As shown by some examples with real and synthetic data, the proposed approach presents key advantages with respect to several existing methods of a different nature, which may fail to find the optimal estimate, may not allow one to establish the optimality of the found estimate, or may require a larger computational burden.</p>
	]]></content:encoded>

	<dc:title>Economical, Optimal and Uncertain Multiple-View L2 Triangulation via LMIs</dc:title>
			<dc:creator>Graziano Chesi</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070222</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-07-05</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-07-05</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>222</prism:startingPage>
		<prism:doi>10.3390/bdcc10070222</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/222</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/221">

	<title>BDCC, Vol. 10, Pages 221: An Information-Theoretic Framework for Characterizing Interaction-Order Diversity in Temporal Hypergraphs</title>
	<link>https://www.mdpi.com/2504-2289/10/7/221</link>
	<description>The proliferation of large-scale interaction datasets, from scientific collaboration networks and legislative records to online communication platforms, has made the analysis of group-based, time-varying systems one of the central challenges of modern data analytics. Hypergraphs provide a natural formalism for such systems, where interactions involve arbitrary groups of agents rather than isolated pairs, and temporal hypergraphs extend this to sequential data by capturing how group interactions evolve over time. Yet quantifying how complex, predictable, or volatile this evolution is remains an open problem: existing entropy-based measures either operate on pairwise projections and thus discard multi-way dependencies or are not naturally defined for varying hyperedge sizes. In this paper, we propose an information&amp;amp;ndash;theoretic framework for characterizing how the diversity of interaction orders in a temporal hypergraph evolves over time. We introduce the hyperedge-size distribution entropy of a snapshot and, building on the theory of entropy rates for stochastic processes, we define the temporal hypergraph entropy rate as a principled, dataset-agnostic measure of the average diversity of interaction orders exhibited by the snapshot sequence over time. We further equip the framework with a bias-corrected sliding-window estimator and a lightweight change-point detector, assembling a complete pipeline that runs in time linear in the total number of hyperedges and requires no node alignment across datasets or snapshots. We prove that the measure collapses to zero under clique expansion, demonstrating that it captures interaction-order information that is discarded by the standard size-blind pairwise projection. Experiments on six small and large publicly available benchmark datasets show that the entropy rate spans 1.60 bits across domains, detects unsupervised structural change points, and discriminates between structurally distinct interaction cultures even within the same domain. Our framework is computationally lightweight and applicable to any dataset that can be represented as a temporal sequence of hypergraphs, paving the way for practical, scalable, interaction-order-aware analysis of large-scale higher-order temporal data.</description>
	<pubDate>2026-07-03</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 221: An Information-Theoretic Framework for Characterizing Interaction-Order Diversity in Temporal Hypergraphs</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/221">doi: 10.3390/bdcc10070221</a></p>
	<p>Authors:
		Francesco Cauteruccio
		</p>
	<p>The proliferation of large-scale interaction datasets, from scientific collaboration networks and legislative records to online communication platforms, has made the analysis of group-based, time-varying systems one of the central challenges of modern data analytics. Hypergraphs provide a natural formalism for such systems, where interactions involve arbitrary groups of agents rather than isolated pairs, and temporal hypergraphs extend this to sequential data by capturing how group interactions evolve over time. Yet quantifying how complex, predictable, or volatile this evolution is remains an open problem: existing entropy-based measures either operate on pairwise projections and thus discard multi-way dependencies or are not naturally defined for varying hyperedge sizes. In this paper, we propose an information&amp;amp;ndash;theoretic framework for characterizing how the diversity of interaction orders in a temporal hypergraph evolves over time. We introduce the hyperedge-size distribution entropy of a snapshot and, building on the theory of entropy rates for stochastic processes, we define the temporal hypergraph entropy rate as a principled, dataset-agnostic measure of the average diversity of interaction orders exhibited by the snapshot sequence over time. We further equip the framework with a bias-corrected sliding-window estimator and a lightweight change-point detector, assembling a complete pipeline that runs in time linear in the total number of hyperedges and requires no node alignment across datasets or snapshots. We prove that the measure collapses to zero under clique expansion, demonstrating that it captures interaction-order information that is discarded by the standard size-blind pairwise projection. Experiments on six small and large publicly available benchmark datasets show that the entropy rate spans 1.60 bits across domains, detects unsupervised structural change points, and discriminates between structurally distinct interaction cultures even within the same domain. Our framework is computationally lightweight and applicable to any dataset that can be represented as a temporal sequence of hypergraphs, paving the way for practical, scalable, interaction-order-aware analysis of large-scale higher-order temporal data.</p>
	]]></content:encoded>

	<dc:title>An Information-Theoretic Framework for Characterizing Interaction-Order Diversity in Temporal Hypergraphs</dc:title>
			<dc:creator>Francesco Cauteruccio</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070221</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-07-03</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-07-03</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>221</prism:startingPage>
		<prism:doi>10.3390/bdcc10070221</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/221</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/220">

	<title>BDCC, Vol. 10, Pages 220: Rank-Adaptive Bayesian Tensor Ring Completion for Low-Altitude 5D Radio Environment Map Construction</title>
	<link>https://www.mdpi.com/2504-2289/10/7/220</link>
	<description>The rapid development of the low-altitude economy demands comprehensive electromagnetic spectrum awareness. However, constructing a comprehensive radio environment map (REM) in this scenario is challenging, as spectrum sensing data collected by unmanned aerial vehicles (UAVs) in complex low-altitude environments is typically sparse, fragmented, and non-uniformly distributed across the high-dimensional space of time, frequency, and 3D space. To address these issues, this study proposes a rank-adaptive Bayesian tensor ring completion (Ra-BTRC) framework. The method models the low-altitude electromagnetic environment as a unified five-dimensional (5D) spectrum tensor. It then employs tensor ring (TR) decomposition to capture latent high-order correlations across all dimensions. To overcome the sensitivity of conventional TR methods to predefined ranks, Ra-BTRC introduces sparsity-inducing priors on the TR core factors, enabling variational Bayesian inference to learn observation uncertainty and infer effective TR ranks from sparse measurements without manually fixing the TR rank. Simulations demonstrate that Ra-BTRC significantly outperforms existing TR-based baselines, achieving more than 10 dB MMSE improvement at a 5% sampling rate while accurately recovering local spectrum structures and temporal dynamics. The proposed approach provides a robust and scalable solution for reliable global low-altitude spectrum cognition under stringent sensing budgets.</description>
	<pubDate>2026-07-03</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 220: Rank-Adaptive Bayesian Tensor Ring Completion for Low-Altitude 5D Radio Environment Map Construction</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/220">doi: 10.3390/bdcc10070220</a></p>
	<p>Authors:
		Ying Wang
		Zhuo Sun
		Hao Ma
		</p>
	<p>The rapid development of the low-altitude economy demands comprehensive electromagnetic spectrum awareness. However, constructing a comprehensive radio environment map (REM) in this scenario is challenging, as spectrum sensing data collected by unmanned aerial vehicles (UAVs) in complex low-altitude environments is typically sparse, fragmented, and non-uniformly distributed across the high-dimensional space of time, frequency, and 3D space. To address these issues, this study proposes a rank-adaptive Bayesian tensor ring completion (Ra-BTRC) framework. The method models the low-altitude electromagnetic environment as a unified five-dimensional (5D) spectrum tensor. It then employs tensor ring (TR) decomposition to capture latent high-order correlations across all dimensions. To overcome the sensitivity of conventional TR methods to predefined ranks, Ra-BTRC introduces sparsity-inducing priors on the TR core factors, enabling variational Bayesian inference to learn observation uncertainty and infer effective TR ranks from sparse measurements without manually fixing the TR rank. Simulations demonstrate that Ra-BTRC significantly outperforms existing TR-based baselines, achieving more than 10 dB MMSE improvement at a 5% sampling rate while accurately recovering local spectrum structures and temporal dynamics. The proposed approach provides a robust and scalable solution for reliable global low-altitude spectrum cognition under stringent sensing budgets.</p>
	]]></content:encoded>

	<dc:title>Rank-Adaptive Bayesian Tensor Ring Completion for Low-Altitude 5D Radio Environment Map Construction</dc:title>
			<dc:creator>Ying Wang</dc:creator>
			<dc:creator>Zhuo Sun</dc:creator>
			<dc:creator>Hao Ma</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070220</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-07-03</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-07-03</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>220</prism:startingPage>
		<prism:doi>10.3390/bdcc10070220</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/220</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/219">

	<title>BDCC, Vol. 10, Pages 219: A Deterministic State Machine Orchestrator with Local LLM Improving Personalized Education Quality Through Interactive Virtual Tutoring Agent with KPI Tracking</title>
	<link>https://www.mdpi.com/2504-2289/10/7/219</link>
	<description>Artificial intelligence is rapidly changing education. However, many learning chatbots are still reactive tools, which respond to arbitrary questions without leading learners through a meaningful pedagogical journey. This article presents a deterministic state-machine orchestrator coupled with a local large language model and a knowledge-graph-framed tutoring strategy for personalized education. The proposed virtual tutoring agent is designed to combine the flexibility of conversational AI with the reliability of explicit instructional states, key performance indicator (KPI) tracking, learner profiling, and controlled transitions between explanation, practice, feedback, assessment, and remediation. The system is not meant to replace the teacher, but rather to act as a teaching co-pilot that provides ongoing feedback, personalized learning paths, accessibility, and safer deployment by processing data locally. The study also presents a compact interview-based evaluation framework and statistical analysis of user perceptions across interactivity, individuality, proactivity, security, accessibility, gamification, and global preference for educational agents over classical chatbots. The findings show that learners appreciate personalized and interactive support and that proactivity is the key feature that distinguishes an educational agent from a regular chatbot. With this article we argue that deterministic orchestration can help make AI tutoring more transparent, controllable, and ethically fit for real learning contexts. Finally, it discusses privacy, educational value, limitations and future improvements to be made before the large-scale adoption of such systems.</description>
	<pubDate>2026-07-03</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 219: A Deterministic State Machine Orchestrator with Local LLM Improving Personalized Education Quality Through Interactive Virtual Tutoring Agent with KPI Tracking</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/219">doi: 10.3390/bdcc10070219</a></p>
	<p>Authors:
		Smail Tigani
		</p>
	<p>Artificial intelligence is rapidly changing education. However, many learning chatbots are still reactive tools, which respond to arbitrary questions without leading learners through a meaningful pedagogical journey. This article presents a deterministic state-machine orchestrator coupled with a local large language model and a knowledge-graph-framed tutoring strategy for personalized education. The proposed virtual tutoring agent is designed to combine the flexibility of conversational AI with the reliability of explicit instructional states, key performance indicator (KPI) tracking, learner profiling, and controlled transitions between explanation, practice, feedback, assessment, and remediation. The system is not meant to replace the teacher, but rather to act as a teaching co-pilot that provides ongoing feedback, personalized learning paths, accessibility, and safer deployment by processing data locally. The study also presents a compact interview-based evaluation framework and statistical analysis of user perceptions across interactivity, individuality, proactivity, security, accessibility, gamification, and global preference for educational agents over classical chatbots. The findings show that learners appreciate personalized and interactive support and that proactivity is the key feature that distinguishes an educational agent from a regular chatbot. With this article we argue that deterministic orchestration can help make AI tutoring more transparent, controllable, and ethically fit for real learning contexts. Finally, it discusses privacy, educational value, limitations and future improvements to be made before the large-scale adoption of such systems.</p>
	]]></content:encoded>

	<dc:title>A Deterministic State Machine Orchestrator with Local LLM Improving Personalized Education Quality Through Interactive Virtual Tutoring Agent with KPI Tracking</dc:title>
			<dc:creator>Smail Tigani</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070219</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-07-03</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-07-03</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>219</prism:startingPage>
		<prism:doi>10.3390/bdcc10070219</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/219</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/218">

	<title>BDCC, Vol. 10, Pages 218: Convolutive Kernel-Guarded Spiking Neural P Systems for Local Feature Computation</title>
	<link>https://www.mdpi.com/2504-2289/10/7/218</link>
	<description>Spiking Neural P systems provide a rule-based model of distributed computation inspired by membrane computing, while kernel P systems use guarded transformations and structured control of rule applicability. This paper introduces Convolutive Kernel-Guarded Spiking Neural P systems (CK-SNP systems), a formal and trainable framework in which spike-rule applicability may depend on local kernel responses computed over ordered neighborhoods of spike multiplicities. The proposed model provides a general mechanism for local feature computation, combining explicit operational semantics with kernel-based predicates that can be fixed, selected, or embedded in trainable realizations. We define the syntax and transition semantics of the model, relate the construction to delay-free extended Spiking Neural P systems and kernel P systems under stated assumptions, and present a reproducible instantiation for electrocardiographic beat classification under a patient-independent protocol. The empirical study illustrates how CK&amp;amp;ndash;SN P local responses can be combined with RR, Gaussian, and Fourier descriptors and evaluated with classical and neural classifiers. Overall, the study clarifies both the formal role of guarded local computation and its practical use as an interpretable feature-generation mechanism.</description>
	<pubDate>2026-07-03</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 218: Convolutive Kernel-Guarded Spiking Neural P Systems for Local Feature Computation</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/218">doi: 10.3390/bdcc10070218</a></p>
	<p>Authors:
		Doru Constantin
		Costel Bălcău
		</p>
	<p>Spiking Neural P systems provide a rule-based model of distributed computation inspired by membrane computing, while kernel P systems use guarded transformations and structured control of rule applicability. This paper introduces Convolutive Kernel-Guarded Spiking Neural P systems (CK-SNP systems), a formal and trainable framework in which spike-rule applicability may depend on local kernel responses computed over ordered neighborhoods of spike multiplicities. The proposed model provides a general mechanism for local feature computation, combining explicit operational semantics with kernel-based predicates that can be fixed, selected, or embedded in trainable realizations. We define the syntax and transition semantics of the model, relate the construction to delay-free extended Spiking Neural P systems and kernel P systems under stated assumptions, and present a reproducible instantiation for electrocardiographic beat classification under a patient-independent protocol. The empirical study illustrates how CK&amp;amp;ndash;SN P local responses can be combined with RR, Gaussian, and Fourier descriptors and evaluated with classical and neural classifiers. Overall, the study clarifies both the formal role of guarded local computation and its practical use as an interpretable feature-generation mechanism.</p>
	]]></content:encoded>

	<dc:title>Convolutive Kernel-Guarded Spiking Neural P Systems for Local Feature Computation</dc:title>
			<dc:creator>Doru Constantin</dc:creator>
			<dc:creator>Costel Bălcău</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070218</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-07-03</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-07-03</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>218</prism:startingPage>
		<prism:doi>10.3390/bdcc10070218</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/218</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/217">

	<title>BDCC, Vol. 10, Pages 217: Extracting Composition Expression Patterns from Materials Science Patent Documents Using SEP-Tags</title>
	<link>https://www.mdpi.com/2504-2289/10/7/217</link>
	<description>Extracting composition expressions from materials science patent documents is essential for patent document searches. Composition expressions describing a single unit of elements and quantities (e.g., &amp;amp;ldquo;Al: 0.02% or more and 0.08% or less&amp;amp;rdquo;) tend to appear clustered together. In such cases, researchers in the field of materials science who conduct patent searches have found that boundary-indicating phrases are effective for searching. However, there are no concrete examples that have implemented this approach, and the validity of this approach has not been evaluated to date. In this paper, we propose a Separator Tag (SEP-tag) framework as an explicit boundary for composition expressions in named entity recognition labels. This allows the named entity recognition model to simultaneously perform entity recognition and pattern boundary learning within a single end-to-end process. Furthermore, we propose a four-axis evaluation framework that extends the conventional single-entity F1 score to evaluate named entity recognition models using SEP-tags. (1) Entity F1 score excluding structural tags, (2) Exact match rate for correct spans, (3) Predicted span pattern extraction F1 score, (4) Pattern extraction F1 score. We conducted evaluations using RoBERTa-base and BERT-base-Japanese on materials science patent datasets in English (10,166 sentences) and Japanese (975 sentences). Experimental results show that training the model with SEP-tags improved the span exact match rate on the English dataset by approximately 59.72 percentage points (from 15.95% to 75.67%), and reduced false positives in pattern extraction to 1/117 (F1 score: 0.0784 &amp;amp;rarr; 0.8503). In the Japanese dataset, false positives were reduced to 1/123 (F1 score: 0.0361 &amp;amp;rarr; 0.4877). For both languages, the entity F1 score was equivalent to that of the model without SEP-tags (English: |&amp;amp;Delta;F1|&amp;amp;lt;0.002, Japanese: |&amp;amp;Delta;F1|&amp;amp;lt;0.001), with no significant difference found for any of the 13 labels. These results demonstrate that explicit structural boundary tokens are highly effective for extracting composition expression patterns in domain-specific named entity recognition.</description>
	<pubDate>2026-07-03</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 217: Extracting Composition Expression Patterns from Materials Science Patent Documents Using SEP-Tags</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/217">doi: 10.3390/bdcc10070217</a></p>
	<p>Authors:
		Toshihiko Sakai
		Nobuhiko Chiwata
		Tsunenori Mine
		</p>
	<p>Extracting composition expressions from materials science patent documents is essential for patent document searches. Composition expressions describing a single unit of elements and quantities (e.g., &amp;amp;ldquo;Al: 0.02% or more and 0.08% or less&amp;amp;rdquo;) tend to appear clustered together. In such cases, researchers in the field of materials science who conduct patent searches have found that boundary-indicating phrases are effective for searching. However, there are no concrete examples that have implemented this approach, and the validity of this approach has not been evaluated to date. In this paper, we propose a Separator Tag (SEP-tag) framework as an explicit boundary for composition expressions in named entity recognition labels. This allows the named entity recognition model to simultaneously perform entity recognition and pattern boundary learning within a single end-to-end process. Furthermore, we propose a four-axis evaluation framework that extends the conventional single-entity F1 score to evaluate named entity recognition models using SEP-tags. (1) Entity F1 score excluding structural tags, (2) Exact match rate for correct spans, (3) Predicted span pattern extraction F1 score, (4) Pattern extraction F1 score. We conducted evaluations using RoBERTa-base and BERT-base-Japanese on materials science patent datasets in English (10,166 sentences) and Japanese (975 sentences). Experimental results show that training the model with SEP-tags improved the span exact match rate on the English dataset by approximately 59.72 percentage points (from 15.95% to 75.67%), and reduced false positives in pattern extraction to 1/117 (F1 score: 0.0784 &amp;amp;rarr; 0.8503). In the Japanese dataset, false positives were reduced to 1/123 (F1 score: 0.0361 &amp;amp;rarr; 0.4877). For both languages, the entity F1 score was equivalent to that of the model without SEP-tags (English: |&amp;amp;Delta;F1|&amp;amp;lt;0.002, Japanese: |&amp;amp;Delta;F1|&amp;amp;lt;0.001), with no significant difference found for any of the 13 labels. These results demonstrate that explicit structural boundary tokens are highly effective for extracting composition expression patterns in domain-specific named entity recognition.</p>
	]]></content:encoded>

	<dc:title>Extracting Composition Expression Patterns from Materials Science Patent Documents Using SEP-Tags</dc:title>
			<dc:creator>Toshihiko Sakai</dc:creator>
			<dc:creator>Nobuhiko Chiwata</dc:creator>
			<dc:creator>Tsunenori Mine</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070217</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-07-03</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-07-03</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>217</prism:startingPage>
		<prism:doi>10.3390/bdcc10070217</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/217</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/216">

	<title>BDCC, Vol. 10, Pages 216: A Comparative Evaluation of Deep Learning and Rule-Based Models for Sentiment Analysis of 5G/6G Public Discourse on Social Media</title>
	<link>https://www.mdpi.com/2504-2289/10/7/216</link>
	<description>Next-generation communication technologies are increasingly shaping not only network infrastructure and digital services, but also public expectations, risk perceptions, and policy debates. As 5G deployment continues and 6G research accelerates, social responses to communication technologies have arguably become an important dimension of technology adoption, governance, and regulatory decision-making. Social media platforms provide timely and large-scale data sources for public opinion analysis. However, 5G/6G-related discourse often contains domain-specific terminology, technical complaints, and complex emotional expressions, which pose challenges for sentiment analysis. To address this challenge, this study constructs a manually annotated dataset of 1746 5G/6G-related Twitter posts collected across multiple communication-related events. This study aims to provide a domain-specific empirical evaluation of sentiment analysis models by examining classification performance, deployment-oriented inference efficiency, and lightweight domain adaptation. Three sentiment analysis methods are evaluated: twitter&amp;amp;ndash;roberta&amp;amp;ndash;base&amp;amp;ndash;sentiment, bertweet&amp;amp;ndash;base&amp;amp;ndash;sentiment&amp;amp;ndash;analysis, and VADER. In addition, a filtered Amazon Reviews&amp;amp;rsquo;23 subset is used as an external review-style dataset, and a LoRA-based fine-tuning experiment is performed on Twitter-RoBERTa to examine domain adaptability. The results show that pre-trained language models achieve stronger classification performance than the rule-based method, particularly for domain-specific and semantically complex texts. VADER, by contrast, shows high observed efficiency under its CPU-based deployment setting, especially for short-text inference. The LoRA fine-tuned RoBERTa model further improves classification performance on both Twitter and Amazon test sets, indicating that lightweight parameter-efficient adaptation can enhance model robustness in specialized 5G/6G discourse. These findings contribute a domain-specific dataset, a deployment-oriented comparison of sentiment analysis paradigms, and empirical evidence on lightweight domain adaptation for 5G/6G-related public opinion monitoring.</description>
	<pubDate>2026-07-02</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 216: A Comparative Evaluation of Deep Learning and Rule-Based Models for Sentiment Analysis of 5G/6G Public Discourse on Social Media</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/216">doi: 10.3390/bdcc10070216</a></p>
	<p>Authors:
		Hangliang Ding
		Jinfeng Li
		</p>
	<p>Next-generation communication technologies are increasingly shaping not only network infrastructure and digital services, but also public expectations, risk perceptions, and policy debates. As 5G deployment continues and 6G research accelerates, social responses to communication technologies have arguably become an important dimension of technology adoption, governance, and regulatory decision-making. Social media platforms provide timely and large-scale data sources for public opinion analysis. However, 5G/6G-related discourse often contains domain-specific terminology, technical complaints, and complex emotional expressions, which pose challenges for sentiment analysis. To address this challenge, this study constructs a manually annotated dataset of 1746 5G/6G-related Twitter posts collected across multiple communication-related events. This study aims to provide a domain-specific empirical evaluation of sentiment analysis models by examining classification performance, deployment-oriented inference efficiency, and lightweight domain adaptation. Three sentiment analysis methods are evaluated: twitter&amp;amp;ndash;roberta&amp;amp;ndash;base&amp;amp;ndash;sentiment, bertweet&amp;amp;ndash;base&amp;amp;ndash;sentiment&amp;amp;ndash;analysis, and VADER. In addition, a filtered Amazon Reviews&amp;amp;rsquo;23 subset is used as an external review-style dataset, and a LoRA-based fine-tuning experiment is performed on Twitter-RoBERTa to examine domain adaptability. The results show that pre-trained language models achieve stronger classification performance than the rule-based method, particularly for domain-specific and semantically complex texts. VADER, by contrast, shows high observed efficiency under its CPU-based deployment setting, especially for short-text inference. The LoRA fine-tuned RoBERTa model further improves classification performance on both Twitter and Amazon test sets, indicating that lightweight parameter-efficient adaptation can enhance model robustness in specialized 5G/6G discourse. These findings contribute a domain-specific dataset, a deployment-oriented comparison of sentiment analysis paradigms, and empirical evidence on lightweight domain adaptation for 5G/6G-related public opinion monitoring.</p>
	]]></content:encoded>

	<dc:title>A Comparative Evaluation of Deep Learning and Rule-Based Models for Sentiment Analysis of 5G/6G Public Discourse on Social Media</dc:title>
			<dc:creator>Hangliang Ding</dc:creator>
			<dc:creator>Jinfeng Li</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070216</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-07-02</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-07-02</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>216</prism:startingPage>
		<prism:doi>10.3390/bdcc10070216</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/216</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/215">

	<title>BDCC, Vol. 10, Pages 215: Beyond Material Flow with Cognitive Waste Theory: Formalizing the Ninth Waste of Lean Manufacturing Through Quantitative Models of Cognitive Inefficiency</title>
	<link>https://www.mdpi.com/2504-2289/10/7/215</link>
	<description>Lean manufacturing has historically focused on eliminating waste from physical production processes; however, increasing digitalization has shifted a substantial portion of operational effort toward information processing and decision making. Existing Lean frameworks lack formal mechanisms to model and quantify inefficiencies arising within these cognitive processes. This paper introduces Cognitive Waste Theory, a mathematical extension of Lean manufacturing that defines cognitive inefficiency as a distinct form of operational waste. Cognitive waste is conceptualized as non-value-adding mental effort generated by misaligned information flow, task structure, and organizational learning dynamics. The framework decomposes cognitive waste into five analytically separable categories: Information Overload, Context Switching, Knowledge Fragmentation, Cognitive Load, and Learning Lag, each expressed through formal mathematical representations grounded in cognitive and operations theory. To enable quantitative assessment, the study proposes normalized waste functions and develops two composite indices: the Cognitive Efficiency Index (CEI), capturing the ratio of effective decision output to cognitive load, and Information Flow Efficiency (IFE), structured analogously to Overall Equipment Effectiveness. Furthermore, classical Lean instruments are reformulated for analytical application in the cognitive domain through Information Value Stream Mapping and Cognitive 5S. By embedding cognitive constructs within a measurable Lean framework, this work provides an attempt to establish a rigorous foundation for analyzing, comparing, and improving cognitive performance in digitally intensive manufacturing systems.</description>
	<pubDate>2026-07-02</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 215: Beyond Material Flow with Cognitive Waste Theory: Formalizing the Ninth Waste of Lean Manufacturing Through Quantitative Models of Cognitive Inefficiency</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/215">doi: 10.3390/bdcc10070215</a></p>
	<p>Authors:
		Mohammad Shahin
		Mazdak Maghanaki
		F. Frank Chen
		</p>
	<p>Lean manufacturing has historically focused on eliminating waste from physical production processes; however, increasing digitalization has shifted a substantial portion of operational effort toward information processing and decision making. Existing Lean frameworks lack formal mechanisms to model and quantify inefficiencies arising within these cognitive processes. This paper introduces Cognitive Waste Theory, a mathematical extension of Lean manufacturing that defines cognitive inefficiency as a distinct form of operational waste. Cognitive waste is conceptualized as non-value-adding mental effort generated by misaligned information flow, task structure, and organizational learning dynamics. The framework decomposes cognitive waste into five analytically separable categories: Information Overload, Context Switching, Knowledge Fragmentation, Cognitive Load, and Learning Lag, each expressed through formal mathematical representations grounded in cognitive and operations theory. To enable quantitative assessment, the study proposes normalized waste functions and develops two composite indices: the Cognitive Efficiency Index (CEI), capturing the ratio of effective decision output to cognitive load, and Information Flow Efficiency (IFE), structured analogously to Overall Equipment Effectiveness. Furthermore, classical Lean instruments are reformulated for analytical application in the cognitive domain through Information Value Stream Mapping and Cognitive 5S. By embedding cognitive constructs within a measurable Lean framework, this work provides an attempt to establish a rigorous foundation for analyzing, comparing, and improving cognitive performance in digitally intensive manufacturing systems.</p>
	]]></content:encoded>

	<dc:title>Beyond Material Flow with Cognitive Waste Theory: Formalizing the Ninth Waste of Lean Manufacturing Through Quantitative Models of Cognitive Inefficiency</dc:title>
			<dc:creator>Mohammad Shahin</dc:creator>
			<dc:creator>Mazdak Maghanaki</dc:creator>
			<dc:creator>F. Frank Chen</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070215</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-07-02</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-07-02</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>215</prism:startingPage>
		<prism:doi>10.3390/bdcc10070215</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/215</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/214">

	<title>BDCC, Vol. 10, Pages 214: Temporal Patterns of Advanced Shot-Quality Metrics in Elite Men&amp;rsquo;s and Women&amp;rsquo;s European Football</title>
	<link>https://www.mdpi.com/2504-2289/10/7/214</link>
	<description>The temporal dynamics of shot quality in elite football remain poorly understood, despite well-documented declines in physical and technical performance during matches. This study aimed to analyze the evolution of advanced shot metrics across match halves and 15 min intervals in elite men&amp;amp;rsquo;s and women&amp;amp;rsquo;s international competitions. A total of 4074 shots from the UEFA European Championships were examined. To ensure methodological consistency among the three advanced shooting metrics (expected goals, xG; expected shot impact timing, xSIT; and expected goals on target, xGOT), analyses were restricted to shots on target (men: 775; women: 554), as xGOT can only be calculated for on-target attempts. Shot quality was assessed using xG, xSIT and xGOT. Differences between halves were evaluated using the Mann&amp;amp;ndash;Whitney U test, while temporal trends were analyzed through linear mixed-effects models. Results showed no significant differences between halves in shot distribution or quality metrics in either competition (all p &amp;amp;gt; 0.05). Likewise, no significant temporal variations were found across the six match intervals for any metric. Women&amp;amp;rsquo;s football exhibited a largely stable quality of shots on target throughout the match. In men&amp;amp;rsquo;s competitions, although a significant difference in xG was observed (p = 0.04, effect sizes were trivial (d = &amp;amp;minus;0.17), and no consistent patterns emerged in xSIT or xGOT. No sex-related differences were observed. Overall, shot quality remained stable despite match progression, suggesting that changes in goal frequency are not driven by variations in the intrinsic quality of shooting opportunities. Therefore, these findings suggest that optimizing the contextual conditions that facilitate the creation of shooting opportunities may be as important as, or more important than, focusing solely on shooting execution.</description>
	<pubDate>2026-07-01</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 214: Temporal Patterns of Advanced Shot-Quality Metrics in Elite Men&amp;rsquo;s and Women&amp;rsquo;s European Football</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/214">doi: 10.3390/bdcc10070214</a></p>
	<p>Authors:
		Blanca De-la-Cruz-Torres
		Anselmo Ruiz-de-Alarcón-Quintero
		Miguel Navarro-Castro
		</p>
	<p>The temporal dynamics of shot quality in elite football remain poorly understood, despite well-documented declines in physical and technical performance during matches. This study aimed to analyze the evolution of advanced shot metrics across match halves and 15 min intervals in elite men&amp;amp;rsquo;s and women&amp;amp;rsquo;s international competitions. A total of 4074 shots from the UEFA European Championships were examined. To ensure methodological consistency among the three advanced shooting metrics (expected goals, xG; expected shot impact timing, xSIT; and expected goals on target, xGOT), analyses were restricted to shots on target (men: 775; women: 554), as xGOT can only be calculated for on-target attempts. Shot quality was assessed using xG, xSIT and xGOT. Differences between halves were evaluated using the Mann&amp;amp;ndash;Whitney U test, while temporal trends were analyzed through linear mixed-effects models. Results showed no significant differences between halves in shot distribution or quality metrics in either competition (all p &amp;amp;gt; 0.05). Likewise, no significant temporal variations were found across the six match intervals for any metric. Women&amp;amp;rsquo;s football exhibited a largely stable quality of shots on target throughout the match. In men&amp;amp;rsquo;s competitions, although a significant difference in xG was observed (p = 0.04, effect sizes were trivial (d = &amp;amp;minus;0.17), and no consistent patterns emerged in xSIT or xGOT. No sex-related differences were observed. Overall, shot quality remained stable despite match progression, suggesting that changes in goal frequency are not driven by variations in the intrinsic quality of shooting opportunities. Therefore, these findings suggest that optimizing the contextual conditions that facilitate the creation of shooting opportunities may be as important as, or more important than, focusing solely on shooting execution.</p>
	]]></content:encoded>

	<dc:title>Temporal Patterns of Advanced Shot-Quality Metrics in Elite Men&amp;amp;rsquo;s and Women&amp;amp;rsquo;s European Football</dc:title>
			<dc:creator>Blanca De-la-Cruz-Torres</dc:creator>
			<dc:creator>Anselmo Ruiz-de-Alarcón-Quintero</dc:creator>
			<dc:creator>Miguel Navarro-Castro</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070214</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-07-01</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-07-01</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>214</prism:startingPage>
		<prism:doi>10.3390/bdcc10070214</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/214</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/213">

	<title>BDCC, Vol. 10, Pages 213: Prompt-Structured Priors for Causal Graph Modeling in Career Growth Path Planning: A Reproducible Simulation Benchmark with Public-Data Anchoring</title>
	<link>https://www.mdpi.com/2504-2289/10/7/213</link>
	<description>Career growth path planning is still dominated by statistical association models that summarize historical transitions but do not explicitly represent the causal mechanisms linking capability development, project exposure, policy support, performance improvement, and promotion outcomes. This study develops a reproducible simulation benchmark for evaluating whether prompt-structured priors, when coupled with dual validation, can help assemble intervention-ready career causal graphs. A structural causal model (SCM) first generated 20,000 synthetic career trajectories with known ground-truth dependencies among ten variables, including education, experience, training hours, certification, project exposure, performance, and promotion. Four prompt families-zero-shot, few-shot, Chain-of-Thought (CoT), and CoT plus schema constraints-were instantiated through a controlled prompt-response emulator so that prompt structure could be studied independently of vendor-specific model drift. The emulator gradients should therefore be read as literature-informed design assumptions about structured prompting rather than as empirical measurements from any named production LLM. Candidate edges were subsequently refined by data validation and expert-proxy domain rules. In the main 30-run benchmark, the best prompt-only setting (CoT plus schema) achieved an F1-score of 0.842, while the proposed hybrid method achieved an F1-score of 0.959 and an intervention-effect mean absolute error of 0.0046. Run-wise confidence intervals and approximate significance checks further indicated that the hybrid workflow materially outperformed the prompt-only variants under the benchmark protocol. A public employee-promotion dataset (N= 54,808) was further used as an external plausibility anchor, where KPI attainment, awards, previous ratings, training score, and length of service were all positively associated with promotion. The results indicate that prompt-structured priors can be useful as a transparent proposal-and-validation mechanism, but not as a substitute for direct validation on real LLMs, matched comparisons with standard causal-discovery baselines, or real HR deployment settings. Accordingly, the central aim is a domain-specific methodological benchmark for testing prompt-structured proposal mechanisms in career-growth causal modeling, rather than a claim of standalone LLM causal discovery or a universal benchmark for every causal-discovery setting.</description>
	<pubDate>2026-06-30</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 213: Prompt-Structured Priors for Causal Graph Modeling in Career Growth Path Planning: A Reproducible Simulation Benchmark with Public-Data Anchoring</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/213">doi: 10.3390/bdcc10070213</a></p>
	<p>Authors:
		Yuhan Xie
		Fang Tang
		Yongkang Zhu
		Ming Li
		Feng Yao
		</p>
	<p>Career growth path planning is still dominated by statistical association models that summarize historical transitions but do not explicitly represent the causal mechanisms linking capability development, project exposure, policy support, performance improvement, and promotion outcomes. This study develops a reproducible simulation benchmark for evaluating whether prompt-structured priors, when coupled with dual validation, can help assemble intervention-ready career causal graphs. A structural causal model (SCM) first generated 20,000 synthetic career trajectories with known ground-truth dependencies among ten variables, including education, experience, training hours, certification, project exposure, performance, and promotion. Four prompt families-zero-shot, few-shot, Chain-of-Thought (CoT), and CoT plus schema constraints-were instantiated through a controlled prompt-response emulator so that prompt structure could be studied independently of vendor-specific model drift. The emulator gradients should therefore be read as literature-informed design assumptions about structured prompting rather than as empirical measurements from any named production LLM. Candidate edges were subsequently refined by data validation and expert-proxy domain rules. In the main 30-run benchmark, the best prompt-only setting (CoT plus schema) achieved an F1-score of 0.842, while the proposed hybrid method achieved an F1-score of 0.959 and an intervention-effect mean absolute error of 0.0046. Run-wise confidence intervals and approximate significance checks further indicated that the hybrid workflow materially outperformed the prompt-only variants under the benchmark protocol. A public employee-promotion dataset (N= 54,808) was further used as an external plausibility anchor, where KPI attainment, awards, previous ratings, training score, and length of service were all positively associated with promotion. The results indicate that prompt-structured priors can be useful as a transparent proposal-and-validation mechanism, but not as a substitute for direct validation on real LLMs, matched comparisons with standard causal-discovery baselines, or real HR deployment settings. Accordingly, the central aim is a domain-specific methodological benchmark for testing prompt-structured proposal mechanisms in career-growth causal modeling, rather than a claim of standalone LLM causal discovery or a universal benchmark for every causal-discovery setting.</p>
	]]></content:encoded>

	<dc:title>Prompt-Structured Priors for Causal Graph Modeling in Career Growth Path Planning: A Reproducible Simulation Benchmark with Public-Data Anchoring</dc:title>
			<dc:creator>Yuhan Xie</dc:creator>
			<dc:creator>Fang Tang</dc:creator>
			<dc:creator>Yongkang Zhu</dc:creator>
			<dc:creator>Ming Li</dc:creator>
			<dc:creator>Feng Yao</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070213</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-06-30</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-06-30</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>213</prism:startingPage>
		<prism:doi>10.3390/bdcc10070213</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/213</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/212">

	<title>BDCC, Vol. 10, Pages 212: CollectivIA: Two-Pipeline Multilingual Legal RAG for Moroccan Territorial Governance with LLM-Assisted and Regex-Based Chunking</title>
	<link>https://www.mdpi.com/2504-2289/10/7/212</link>
	<description>Retrieval grounding is crucial for high-stakes administrative applications, since large language models remain prone to hallucinations when addressing legal questions. This problem is particularly relevant in Moroccan territorial governance, where official legislative PDFs have highly heterogeneous digital quality, user interactions often occur in Moroccan Darija, and the legal corpus is bilingual Arabic&amp;amp;ndash;French. This paper presents CollectivIA, a multilingual Retrieval-Augmented Generation system implemented for Moroccan territorial governance law. The system supports queries in French, Arabic, and Moroccan Darija and indexes 2272 article-level segments from sixteen official legislative documents. We compare two end-to-end retrieval pipelines: an LLM-assisted semantic chunking architecture using Gemini and ChromaDB and a regex-based chunking architecture using FAISS. Based on an expanded multilingual benchmark of 150 legal queries, with 50 queries per language group, the LLM-assisted pipeline achieved higher RAGAS scores than the regex-based pipeline, particularly improving Context Precision from 0.315 to 0.818. The multimodal Vision fallback successfully recovered 456 articles, which remained inaccessible under the regex-based pipeline. Overall, the LLM-assisted pipeline yielded legal boundaries with greater coherence and retrieved contexts with higher focus, while the regex-based design maintained a broader source diversity. These results suggest that LLM-assisted semantic chunking with multimodal fallback is a promising approach to enhance multilingual legal RAG over heterogeneous Moroccan legal corpora.</description>
	<pubDate>2026-06-29</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 212: CollectivIA: Two-Pipeline Multilingual Legal RAG for Moroccan Territorial Governance with LLM-Assisted and Regex-Based Chunking</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/212">doi: 10.3390/bdcc10070212</a></p>
	<p>Authors:
		Firiel Zouak
		Omar El Beqqali
		Jamal Riffi
		</p>
	<p>Retrieval grounding is crucial for high-stakes administrative applications, since large language models remain prone to hallucinations when addressing legal questions. This problem is particularly relevant in Moroccan territorial governance, where official legislative PDFs have highly heterogeneous digital quality, user interactions often occur in Moroccan Darija, and the legal corpus is bilingual Arabic&amp;amp;ndash;French. This paper presents CollectivIA, a multilingual Retrieval-Augmented Generation system implemented for Moroccan territorial governance law. The system supports queries in French, Arabic, and Moroccan Darija and indexes 2272 article-level segments from sixteen official legislative documents. We compare two end-to-end retrieval pipelines: an LLM-assisted semantic chunking architecture using Gemini and ChromaDB and a regex-based chunking architecture using FAISS. Based on an expanded multilingual benchmark of 150 legal queries, with 50 queries per language group, the LLM-assisted pipeline achieved higher RAGAS scores than the regex-based pipeline, particularly improving Context Precision from 0.315 to 0.818. The multimodal Vision fallback successfully recovered 456 articles, which remained inaccessible under the regex-based pipeline. Overall, the LLM-assisted pipeline yielded legal boundaries with greater coherence and retrieved contexts with higher focus, while the regex-based design maintained a broader source diversity. These results suggest that LLM-assisted semantic chunking with multimodal fallback is a promising approach to enhance multilingual legal RAG over heterogeneous Moroccan legal corpora.</p>
	]]></content:encoded>

	<dc:title>CollectivIA: Two-Pipeline Multilingual Legal RAG for Moroccan Territorial Governance with LLM-Assisted and Regex-Based Chunking</dc:title>
			<dc:creator>Firiel Zouak</dc:creator>
			<dc:creator>Omar El Beqqali</dc:creator>
			<dc:creator>Jamal Riffi</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070212</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-06-29</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-06-29</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>212</prism:startingPage>
		<prism:doi>10.3390/bdcc10070212</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/212</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/211">

	<title>BDCC, Vol. 10, Pages 211: The Fragility of Phishing Detection Models: Evidence from Cross-Corpus Transfer, Prevalence Shift, Artifact Learning, and Evasion Risk</title>
	<link>https://www.mdpi.com/2504-2289/10/7/211</link>
	<description>Phishing detection models often report strong benchmark performance, yet their reliability under realistic deployment conditions remains uncertain. This study examines this problem by investigating three failure modes of cross-dataset phishing email detection: corpus generalization failure, asymmetric prevalence-shift failure, and artifact-driven spurious learning. Using six public email corpora, CEAS_08, Enron, Ling, Nazario, Nigerian Fraud, and SpamAssassin, the study evaluates Term Frequency (TF) and Inverse Document Frequency (IDF)-based Logistic Regression and Linear Support Vector Classifier (SVC) models across pooled baseline testing, single-corpus cross-dataset transfer, leave-one-corpus-out pooled training, prevalence-shift simulation, training prevalence manipulation, dataset-identification analysis, top-feature inspection, artifact-removal ablation, and targeted feature-sensitivity masking. The findings show that single-corpus models are unstable under cross-dataset transfer, with F1-scores varying substantially across source&amp;amp;ndash;target combinations. In contrast, leave-one-corpus-out pooled training improves robustness, with Logistic Regression achieving sustained F1-scores between 0.8201 and 0.8994, and Linear SVC achieving F1-scores between 0.7607 and 0.8910 across unseen corpora. Prevalence-shift experiments reveal that failure is asymmetric and threshold-dependent. High-prevalence-trained models maintain high recall under fixed thresholds but suffer sharp recall degradation when operational alert-budget constraints are imposed. Conversely, low-prevalence-trained models become overly conservative in high-threat environments, producing high precision but substantially lower recall and poorer calibration. Artifact analyses further show that source corpus identity is highly learnable, with dataset-identification accuracy reaching 0.9722 for Logistic Regression and 0.9806 for Linear SVC. Top-feature and masking analyses indicate that models rely partly on corpus markers, date tokens, URL/domain terms, headers, and other artifact-like features rather than only general phishing indicators. The study contributes a deployment-aware and adversary-aware evaluation framework for phishing detection. It shows that benchmark accuracy alone is insufficient for assessing real-world robustness and that reliable phishing detection requires cross-corpus validation, prevalence-aware thresholding, and systematic testing for artifact-driven spurious learning.</description>
	<pubDate>2026-06-29</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 211: The Fragility of Phishing Detection Models: Evidence from Cross-Corpus Transfer, Prevalence Shift, Artifact Learning, and Evasion Risk</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/211">doi: 10.3390/bdcc10070211</a></p>
	<p>Authors:
		Istiaque Bhuiyan
		Tanvir Bhuiyan
		</p>
	<p>Phishing detection models often report strong benchmark performance, yet their reliability under realistic deployment conditions remains uncertain. This study examines this problem by investigating three failure modes of cross-dataset phishing email detection: corpus generalization failure, asymmetric prevalence-shift failure, and artifact-driven spurious learning. Using six public email corpora, CEAS_08, Enron, Ling, Nazario, Nigerian Fraud, and SpamAssassin, the study evaluates Term Frequency (TF) and Inverse Document Frequency (IDF)-based Logistic Regression and Linear Support Vector Classifier (SVC) models across pooled baseline testing, single-corpus cross-dataset transfer, leave-one-corpus-out pooled training, prevalence-shift simulation, training prevalence manipulation, dataset-identification analysis, top-feature inspection, artifact-removal ablation, and targeted feature-sensitivity masking. The findings show that single-corpus models are unstable under cross-dataset transfer, with F1-scores varying substantially across source&amp;amp;ndash;target combinations. In contrast, leave-one-corpus-out pooled training improves robustness, with Logistic Regression achieving sustained F1-scores between 0.8201 and 0.8994, and Linear SVC achieving F1-scores between 0.7607 and 0.8910 across unseen corpora. Prevalence-shift experiments reveal that failure is asymmetric and threshold-dependent. High-prevalence-trained models maintain high recall under fixed thresholds but suffer sharp recall degradation when operational alert-budget constraints are imposed. Conversely, low-prevalence-trained models become overly conservative in high-threat environments, producing high precision but substantially lower recall and poorer calibration. Artifact analyses further show that source corpus identity is highly learnable, with dataset-identification accuracy reaching 0.9722 for Logistic Regression and 0.9806 for Linear SVC. Top-feature and masking analyses indicate that models rely partly on corpus markers, date tokens, URL/domain terms, headers, and other artifact-like features rather than only general phishing indicators. The study contributes a deployment-aware and adversary-aware evaluation framework for phishing detection. It shows that benchmark accuracy alone is insufficient for assessing real-world robustness and that reliable phishing detection requires cross-corpus validation, prevalence-aware thresholding, and systematic testing for artifact-driven spurious learning.</p>
	]]></content:encoded>

	<dc:title>The Fragility of Phishing Detection Models: Evidence from Cross-Corpus Transfer, Prevalence Shift, Artifact Learning, and Evasion Risk</dc:title>
			<dc:creator>Istiaque Bhuiyan</dc:creator>
			<dc:creator>Tanvir Bhuiyan</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070211</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-06-29</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-06-29</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>211</prism:startingPage>
		<prism:doi>10.3390/bdcc10070211</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/211</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/210">

	<title>BDCC, Vol. 10, Pages 210: Leveraging Large Language Models and Object Detection for Automated Knowledge Graph Generation from Industrial Schematics</title>
	<link>https://www.mdpi.com/2504-2289/10/7/210</link>
	<description>Industrial digitalization increasingly requires automated tools capable of extracting structured knowledge from complex engineering documentation, such as Piping and Instrumentation Diagrams (P&amp;amp;amp;IDs). This work proposes an integrated framework that combines object detection and Large Language Models (LLMs) for automated Knowledge Graph (KG) generation. The approach enables the transformation of unstructured P&amp;amp;amp;ID schematics into machine-interpretable representations, supporting data-driven analysis and decision-making. A modular pipeline is developed, including image pre-processing, symbol detection via a YOLO-based model, and identification of semantic relations between schematic elements using LLMs. The proposal also includes the definition of a reference ontology, which is exploited for the construction of the KG, and a diagram dataset designed to test the performance of the object detection model. The KG generation procedure achieves strong results in terms of image reconstruction across a wide set of industrial schematics, while also preserving the semantic integrity and completeness of the original diagrams. The proposed method represents a significant step toward the digitalization of industrial knowledge, bridging traditional engineering documentation and semantic-based technologies.</description>
	<pubDate>2026-06-29</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 210: Leveraging Large Language Models and Object Detection for Automated Knowledge Graph Generation from Industrial Schematics</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/210">doi: 10.3390/bdcc10070210</a></p>
	<p>Authors:
		Federico Lopomo
		Valentina Faraco
		Davide Marche
		Saverio Ieva
		Giuseppe Loseto
		Davide Loconte
		Floriano Scioscia
		Michele Ruta
		</p>
	<p>Industrial digitalization increasingly requires automated tools capable of extracting structured knowledge from complex engineering documentation, such as Piping and Instrumentation Diagrams (P&amp;amp;amp;IDs). This work proposes an integrated framework that combines object detection and Large Language Models (LLMs) for automated Knowledge Graph (KG) generation. The approach enables the transformation of unstructured P&amp;amp;amp;ID schematics into machine-interpretable representations, supporting data-driven analysis and decision-making. A modular pipeline is developed, including image pre-processing, symbol detection via a YOLO-based model, and identification of semantic relations between schematic elements using LLMs. The proposal also includes the definition of a reference ontology, which is exploited for the construction of the KG, and a diagram dataset designed to test the performance of the object detection model. The KG generation procedure achieves strong results in terms of image reconstruction across a wide set of industrial schematics, while also preserving the semantic integrity and completeness of the original diagrams. The proposed method represents a significant step toward the digitalization of industrial knowledge, bridging traditional engineering documentation and semantic-based technologies.</p>
	]]></content:encoded>

	<dc:title>Leveraging Large Language Models and Object Detection for Automated Knowledge Graph Generation from Industrial Schematics</dc:title>
			<dc:creator>Federico Lopomo</dc:creator>
			<dc:creator>Valentina Faraco</dc:creator>
			<dc:creator>Davide Marche</dc:creator>
			<dc:creator>Saverio Ieva</dc:creator>
			<dc:creator>Giuseppe Loseto</dc:creator>
			<dc:creator>Davide Loconte</dc:creator>
			<dc:creator>Floriano Scioscia</dc:creator>
			<dc:creator>Michele Ruta</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070210</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-06-29</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-06-29</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>210</prism:startingPage>
		<prism:doi>10.3390/bdcc10070210</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/210</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/209">

	<title>BDCC, Vol. 10, Pages 209: Leakage-Guarded Next-Window Superchat Prediction from VTuber Live Chat Dynamics</title>
	<link>https://www.mdpi.com/2504-2289/10/7/209</link>
	<description>Predicting near-future monetization in virtual livestreaming remains methodologically challenging because paid-support events are sparse, temporally dependent, and vulnerable to leakage under inappropriate evaluation designs. This study develops a leakage-guarded, window-based machine-learning framework for predicting next-window Superchat occurrence from VTuber live-chat dynamics. Public VTuber live-chat and Superchat logs were reconstructed into non-overlapping five-minute windows, and features were organized into audience activity, member composition, message intensity, donation-state information, and short-horizon dynamic groups. To reduce optimistic bias, the primary evaluation used video-level grouped splitting and compared a strict setting that excluded direct current-window donation-state variables with an extended donation-state-aware setting. HistGradientBoosting achieved the strongest performance. In the strict setting, it reached PR-AUC = 0.899, ROC-AUC = 0.920, F1 = 0.822, and Brier score = 0.171, while the extended setting produced only modest additional gains. Additional zero-chat sensitivity, repeated grouped split, channel-level robustness, graph-proxy baseline, feature-ablation, and calibration analyses supported the stability and interpretability of the framework. The results suggest that next-window Superchat occurrence can be predicted from participation breadth, chat activity, message intensity, and temporally shifted behavioral dynamics under leakage-aware evaluation.</description>
	<pubDate>2026-06-29</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 209: Leakage-Guarded Next-Window Superchat Prediction from VTuber Live Chat Dynamics</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/209">doi: 10.3390/bdcc10070209</a></p>
	<p>Authors:
		Hwan Soo Yu
		Jae-Uk Kim
		Soo Young Cho
		</p>
	<p>Predicting near-future monetization in virtual livestreaming remains methodologically challenging because paid-support events are sparse, temporally dependent, and vulnerable to leakage under inappropriate evaluation designs. This study develops a leakage-guarded, window-based machine-learning framework for predicting next-window Superchat occurrence from VTuber live-chat dynamics. Public VTuber live-chat and Superchat logs were reconstructed into non-overlapping five-minute windows, and features were organized into audience activity, member composition, message intensity, donation-state information, and short-horizon dynamic groups. To reduce optimistic bias, the primary evaluation used video-level grouped splitting and compared a strict setting that excluded direct current-window donation-state variables with an extended donation-state-aware setting. HistGradientBoosting achieved the strongest performance. In the strict setting, it reached PR-AUC = 0.899, ROC-AUC = 0.920, F1 = 0.822, and Brier score = 0.171, while the extended setting produced only modest additional gains. Additional zero-chat sensitivity, repeated grouped split, channel-level robustness, graph-proxy baseline, feature-ablation, and calibration analyses supported the stability and interpretability of the framework. The results suggest that next-window Superchat occurrence can be predicted from participation breadth, chat activity, message intensity, and temporally shifted behavioral dynamics under leakage-aware evaluation.</p>
	]]></content:encoded>

	<dc:title>Leakage-Guarded Next-Window Superchat Prediction from VTuber Live Chat Dynamics</dc:title>
			<dc:creator>Hwan Soo Yu</dc:creator>
			<dc:creator>Jae-Uk Kim</dc:creator>
			<dc:creator>Soo Young Cho</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070209</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-06-29</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-06-29</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>209</prism:startingPage>
		<prism:doi>10.3390/bdcc10070209</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/209</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/208">

	<title>BDCC, Vol. 10, Pages 208: A Distributed Island-Based Feature Selection Framework for IoT Intrusion Detection Systems</title>
	<link>https://www.mdpi.com/2504-2289/10/7/208</link>
	<description>The widespread deployment of Internet of Things (IoT) environments has led to an increasing number of cyberattacks, highlighting the need for efficient and accurate intrusion detection systems. Over the last few decades, Intrusion Detection Systems (IDSs) have been proposed to tackle this challenge. However, IDSs face challenges when dealing with high-dimensional IoT data that include redundant or irrelevant features, which can lead to increased false positives and decreased detection performance. An optimization-based IDS framework is one of the solutions for reducing data dimensionality and improving detection accuracy. However, the serial implementation of this type of IDS suffers from high computational time as the volume of data and its dimensionality increase. In this paper, we propose a scalable distributed island-based feature selection IDS using Apache Spark (version 3.5.6), called DISFS-IDS. DISFS-IDS follows a two-level partitioning strategy&amp;amp;mdash;data and population&amp;amp;mdash;to distribute the workload across worker nodes in order to identify the most informative features while achieving high detection accuracy. Using binary and multiclass IoT datasets, the experimental results demonstrate that DISFS-IDS achieves statistically comparable detection performance to the serial SFOA-based IDS while selecting a smaller subset of features. Moreover, DISFS-IDS provides effective feature reduction and competitive or superior performance compared with Spark-based filter feature selection methods. In the scalability analysis, DISFS-IDS achieves significant speedup as the number of islands increases while maintaining high parallel efficiency.</description>
	<pubDate>2026-06-27</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 208: A Distributed Island-Based Feature Selection Framework for IoT Intrusion Detection Systems</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/208">doi: 10.3390/bdcc10070208</a></p>
	<p>Authors:
		Jamil Al-Sawwa
		Aws A. Magableh
		</p>
	<p>The widespread deployment of Internet of Things (IoT) environments has led to an increasing number of cyberattacks, highlighting the need for efficient and accurate intrusion detection systems. Over the last few decades, Intrusion Detection Systems (IDSs) have been proposed to tackle this challenge. However, IDSs face challenges when dealing with high-dimensional IoT data that include redundant or irrelevant features, which can lead to increased false positives and decreased detection performance. An optimization-based IDS framework is one of the solutions for reducing data dimensionality and improving detection accuracy. However, the serial implementation of this type of IDS suffers from high computational time as the volume of data and its dimensionality increase. In this paper, we propose a scalable distributed island-based feature selection IDS using Apache Spark (version 3.5.6), called DISFS-IDS. DISFS-IDS follows a two-level partitioning strategy&amp;amp;mdash;data and population&amp;amp;mdash;to distribute the workload across worker nodes in order to identify the most informative features while achieving high detection accuracy. Using binary and multiclass IoT datasets, the experimental results demonstrate that DISFS-IDS achieves statistically comparable detection performance to the serial SFOA-based IDS while selecting a smaller subset of features. Moreover, DISFS-IDS provides effective feature reduction and competitive or superior performance compared with Spark-based filter feature selection methods. In the scalability analysis, DISFS-IDS achieves significant speedup as the number of islands increases while maintaining high parallel efficiency.</p>
	]]></content:encoded>

	<dc:title>A Distributed Island-Based Feature Selection Framework for IoT Intrusion Detection Systems</dc:title>
			<dc:creator>Jamil Al-Sawwa</dc:creator>
			<dc:creator>Aws A. Magableh</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070208</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-06-27</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-06-27</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>208</prism:startingPage>
		<prism:doi>10.3390/bdcc10070208</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/208</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/207">

	<title>BDCC, Vol. 10, Pages 207: A Deep Graph Regularized Lp Smooth Semi-Non-Negative Matrix Factorization Method for Image Clustering</title>
	<link>https://www.mdpi.com/2504-2289/10/7/207</link>
	<description>Unsupervised learning often relies on non-negative matrix factorization (NMF) for extracting low-dimensional features. Standard deep NMF models, however, tend to miss complex hierarchical patterns and may warp the intrinsic geometry of high-dimensional data, resulting in solutions that are neither smooth nor stable. To counter these issues, we introduce DGLpSNMF&amp;amp;mdash;a deep graph-regularized Lp smooth semi-NMF&amp;amp;mdash;that explicitly incorporates the data&amp;amp;rsquo;s geometric structure via graph Laplacian regularization and Lp smoothing. The optimization problem is tackled with a forward-backward splitting scheme, and we establish convergence of the generated sequence to a critical point. Experiments on four image benchmarks (JAFFE, Yale, ORL, PIE) demonstrate that DGLpSNMF consistently surpasses several state-of-the-art NMF variants in both accuracy and normalized mutual information.</description>
	<pubDate>2026-06-25</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 207: A Deep Graph Regularized Lp Smooth Semi-Non-Negative Matrix Factorization Method for Image Clustering</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/207">doi: 10.3390/bdcc10070207</a></p>
	<p>Authors:
		Shunli Li
		Mingjun Bai
		Ling Wang
		</p>
	<p>Unsupervised learning often relies on non-negative matrix factorization (NMF) for extracting low-dimensional features. Standard deep NMF models, however, tend to miss complex hierarchical patterns and may warp the intrinsic geometry of high-dimensional data, resulting in solutions that are neither smooth nor stable. To counter these issues, we introduce DGLpSNMF&amp;amp;mdash;a deep graph-regularized Lp smooth semi-NMF&amp;amp;mdash;that explicitly incorporates the data&amp;amp;rsquo;s geometric structure via graph Laplacian regularization and Lp smoothing. The optimization problem is tackled with a forward-backward splitting scheme, and we establish convergence of the generated sequence to a critical point. Experiments on four image benchmarks (JAFFE, Yale, ORL, PIE) demonstrate that DGLpSNMF consistently surpasses several state-of-the-art NMF variants in both accuracy and normalized mutual information.</p>
	]]></content:encoded>

	<dc:title>A Deep Graph Regularized Lp Smooth Semi-Non-Negative Matrix Factorization Method for Image Clustering</dc:title>
			<dc:creator>Shunli Li</dc:creator>
			<dc:creator>Mingjun Bai</dc:creator>
			<dc:creator>Ling Wang</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070207</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-06-25</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-06-25</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>207</prism:startingPage>
		<prism:doi>10.3390/bdcc10070207</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/207</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/206">

	<title>BDCC, Vol. 10, Pages 206: Resource Allocation via Bayesian Optimization in Wasserstein Spaces vs. Semi-Bandit Feedback</title>
	<link>https://www.mdpi.com/2504-2289/10/7/206</link>
	<description>Sequential resource allocation has long been a central problem in operations research, yet ongoing technological developments, particularly in cloud and high-performance computing and in multi-channel marketing, are giving rise to new structural constraints that classical methods were not designed to handle. Semi-Bandit Feedback (SBF) has emerged as the dominant framework for these modern settings. This paper introduces an alternative that recasts the allocation problem within the Bayesian Optimization (BO) paradigm. All three proposed BO algorithms consistently outperform SBF, with BORAwSE showing a particularly clear advantage under time-varying budget settings, while CBO achieves comparable rewards under constant budget conditions. The core methodological contribution is a reformulation in which each candidate allocation is represented as a discrete probability distribution over the available options, making the probability simplex the natural search domain. Grounding the search in this space calls for a geometry that respects the structure of distributions: we adopt the optimal transport (Wasserstein) distance, which allows both the Gaussian process surrogate and the acquisition function to be extended as functionals over the simplex. A further practical advantage of the proposed method is its applicability to problem instances where SBF cannot be used without modification. The approach is evaluated on two case studies: the benchmark computing-resource allocation scenario from the original SBF paper, and a budget allocation problem across marketing channels.</description>
	<pubDate>2026-06-25</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 206: Resource Allocation via Bayesian Optimization in Wasserstein Spaces vs. Semi-Bandit Feedback</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/206">doi: 10.3390/bdcc10070206</a></p>
	<p>Authors:
		Antonio Candelieri
		Francesco Archetti
		Iman Seyedi
		Andrea Ponti
		</p>
	<p>Sequential resource allocation has long been a central problem in operations research, yet ongoing technological developments, particularly in cloud and high-performance computing and in multi-channel marketing, are giving rise to new structural constraints that classical methods were not designed to handle. Semi-Bandit Feedback (SBF) has emerged as the dominant framework for these modern settings. This paper introduces an alternative that recasts the allocation problem within the Bayesian Optimization (BO) paradigm. All three proposed BO algorithms consistently outperform SBF, with BORAwSE showing a particularly clear advantage under time-varying budget settings, while CBO achieves comparable rewards under constant budget conditions. The core methodological contribution is a reformulation in which each candidate allocation is represented as a discrete probability distribution over the available options, making the probability simplex the natural search domain. Grounding the search in this space calls for a geometry that respects the structure of distributions: we adopt the optimal transport (Wasserstein) distance, which allows both the Gaussian process surrogate and the acquisition function to be extended as functionals over the simplex. A further practical advantage of the proposed method is its applicability to problem instances where SBF cannot be used without modification. The approach is evaluated on two case studies: the benchmark computing-resource allocation scenario from the original SBF paper, and a budget allocation problem across marketing channels.</p>
	]]></content:encoded>

	<dc:title>Resource Allocation via Bayesian Optimization in Wasserstein Spaces vs. Semi-Bandit Feedback</dc:title>
			<dc:creator>Antonio Candelieri</dc:creator>
			<dc:creator>Francesco Archetti</dc:creator>
			<dc:creator>Iman Seyedi</dc:creator>
			<dc:creator>Andrea Ponti</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070206</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-06-25</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-06-25</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>206</prism:startingPage>
		<prism:doi>10.3390/bdcc10070206</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/206</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/205">

	<title>BDCC, Vol. 10, Pages 205: Band-Limited Proximal FISTA for Efficient Sparse Harmonic Recovery on MCU</title>
	<link>https://www.mdpi.com/2504-2289/10/7/205</link>
	<description>Compressed sensing (CS) enables signal reconstruction from fewer measurements when the signal is sparse in a transform domain. However, executing &amp;amp;#8467;1-regularized recovery on MCU-class hardware is challenging due to limited compute resources and the cost of repeated forward and adjoint operator evaluations. This paper presents a band-limited proximal variant of FISTA that enforces known spectral support during thresholding, restricting the effective optimization domain without changing the measurement model. We implement a complete CS reconstruction pipeline on an STM32F407 (Cortex-M4) using CMSIS-DSP FFT/IFFT kernels and evaluate it using ECG waveforms acquired through an AD8232 front end as benchmark signals. With M=340 measurements (33% of uniform sampling), the embedded implementation achieves a PRDN of 24.38%, closely matching MATLAB references (CVX: 22.64%, FISTA: 22.39%) under identical hyperparameters. Cycle-accurate profiling shows that FFT/IFFT-based forward/adjoint operators dominate the per-iteration runtime. Under a 60 Hz band-limited setting, the required iterations are reduced from 30 to 16 with an acceptable PRDN, demonstrating a practical trade-off between reconstruction accuracy and computational cost on MCU-class devices.</description>
	<pubDate>2026-06-25</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 205: Band-Limited Proximal FISTA for Efficient Sparse Harmonic Recovery on MCU</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/205">doi: 10.3390/bdcc10070205</a></p>
	<p>Authors:
		Seongho Cho
		Minjung Kim
		Daejin Park
		</p>
	<p>Compressed sensing (CS) enables signal reconstruction from fewer measurements when the signal is sparse in a transform domain. However, executing &amp;amp;#8467;1-regularized recovery on MCU-class hardware is challenging due to limited compute resources and the cost of repeated forward and adjoint operator evaluations. This paper presents a band-limited proximal variant of FISTA that enforces known spectral support during thresholding, restricting the effective optimization domain without changing the measurement model. We implement a complete CS reconstruction pipeline on an STM32F407 (Cortex-M4) using CMSIS-DSP FFT/IFFT kernels and evaluate it using ECG waveforms acquired through an AD8232 front end as benchmark signals. With M=340 measurements (33% of uniform sampling), the embedded implementation achieves a PRDN of 24.38%, closely matching MATLAB references (CVX: 22.64%, FISTA: 22.39%) under identical hyperparameters. Cycle-accurate profiling shows that FFT/IFFT-based forward/adjoint operators dominate the per-iteration runtime. Under a 60 Hz band-limited setting, the required iterations are reduced from 30 to 16 with an acceptable PRDN, demonstrating a practical trade-off between reconstruction accuracy and computational cost on MCU-class devices.</p>
	]]></content:encoded>

	<dc:title>Band-Limited Proximal FISTA for Efficient Sparse Harmonic Recovery on MCU</dc:title>
			<dc:creator>Seongho Cho</dc:creator>
			<dc:creator>Minjung Kim</dc:creator>
			<dc:creator>Daejin Park</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070205</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-06-25</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-06-25</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>205</prism:startingPage>
		<prism:doi>10.3390/bdcc10070205</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/205</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/204">

	<title>BDCC, Vol. 10, Pages 204: Recognition of Acupoints on Human Back Based on Machine Vision and Deep Learning</title>
	<link>https://www.mdpi.com/2504-2289/10/7/204</link>
	<description>Traditional acupoint localization methods rely heavily on manual operation, resulting in high subjectivity and limited accuracy. To improve the precision and stability of acupoint detection, this study integrates machine vision technology with in situ projection to achieve automated recognition and real-time visualization of human acupoints. First, an automatic calibration method based on image processing is proposed for back acupoints. Spinal features are extracted from the blue channel, enhanced using adaptive histogram equalization, and processed through region of interest extraction, minimum-threshold binarization, and morphological operations. Key spinal curve points are then fitted using B&amp;amp;eacute;zier functions. Canny edge detection is used to extract the human silhouette, locate the acromion, and derive the pixel scale of the &amp;amp;ldquo;cun&amp;amp;rdquo; measurement, enabling coordinate computation for 141 back acupoints. In the deep learning component, an improved YOLOv8-Pose model is developed for acupoint localization. Unlike existing methods that use local attention or the original Object Keypoint Similarity (OKS) loss, we introduce two innovations: a non-local attention module for global dependency modeling, and a novel Efficient Object Keypoint Similarity (EOKS) loss function that incorporates geometric constraints&amp;amp;mdash;namely, width, height, and center distance&amp;amp;mdash;in addition to Euclidean distance. A non-local attention mechanism is incorporated into the backbone to enhance global feature extraction, and the EOKS loss function is designed to improve spatiogeometric regression accuracy. An inference mechanism is further introduced to derive the remaining acupoints from 49 detected keypoints; experiments demonstrate that the improved model achieves 95.0% detection accuracy, outperforming the baseline by 2.62%, with an inference time of 14.5 ms. Finally, an in situ projection platform is constructed, combining camera calibration, four-point proportional scaling, and an OpenCV 4.5.4-based interactive interface. The system supports real-time translation, rotation, and scaling, enabling accurate projection of detected acupoints onto the human body.</description>
	<pubDate>2026-06-23</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 204: Recognition of Acupoints on Human Back Based on Machine Vision and Deep Learning</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/204">doi: 10.3390/bdcc10070204</a></p>
	<p>Authors:
		Zhike Zhao
		Linman Song
		Songying Li
		Ruihao Xue
		Peng Li
		</p>
	<p>Traditional acupoint localization methods rely heavily on manual operation, resulting in high subjectivity and limited accuracy. To improve the precision and stability of acupoint detection, this study integrates machine vision technology with in situ projection to achieve automated recognition and real-time visualization of human acupoints. First, an automatic calibration method based on image processing is proposed for back acupoints. Spinal features are extracted from the blue channel, enhanced using adaptive histogram equalization, and processed through region of interest extraction, minimum-threshold binarization, and morphological operations. Key spinal curve points are then fitted using B&amp;amp;eacute;zier functions. Canny edge detection is used to extract the human silhouette, locate the acromion, and derive the pixel scale of the &amp;amp;ldquo;cun&amp;amp;rdquo; measurement, enabling coordinate computation for 141 back acupoints. In the deep learning component, an improved YOLOv8-Pose model is developed for acupoint localization. Unlike existing methods that use local attention or the original Object Keypoint Similarity (OKS) loss, we introduce two innovations: a non-local attention module for global dependency modeling, and a novel Efficient Object Keypoint Similarity (EOKS) loss function that incorporates geometric constraints&amp;amp;mdash;namely, width, height, and center distance&amp;amp;mdash;in addition to Euclidean distance. A non-local attention mechanism is incorporated into the backbone to enhance global feature extraction, and the EOKS loss function is designed to improve spatiogeometric regression accuracy. An inference mechanism is further introduced to derive the remaining acupoints from 49 detected keypoints; experiments demonstrate that the improved model achieves 95.0% detection accuracy, outperforming the baseline by 2.62%, with an inference time of 14.5 ms. Finally, an in situ projection platform is constructed, combining camera calibration, four-point proportional scaling, and an OpenCV 4.5.4-based interactive interface. The system supports real-time translation, rotation, and scaling, enabling accurate projection of detected acupoints onto the human body.</p>
	]]></content:encoded>

	<dc:title>Recognition of Acupoints on Human Back Based on Machine Vision and Deep Learning</dc:title>
			<dc:creator>Zhike Zhao</dc:creator>
			<dc:creator>Linman Song</dc:creator>
			<dc:creator>Songying Li</dc:creator>
			<dc:creator>Ruihao Xue</dc:creator>
			<dc:creator>Peng Li</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070204</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-06-23</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-06-23</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>204</prism:startingPage>
		<prism:doi>10.3390/bdcc10070204</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/204</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/203">

	<title>BDCC, Vol. 10, Pages 203: Transformative Simulation as an Ontology for AI in Health Systems: From Fluent Tools to Coherent Reasoning</title>
	<link>https://www.mdpi.com/2504-2289/10/7/203</link>
	<description>Artificial intelligence (AI) is increasingly applied to healthcare decision-making; however, many persistent patient safety risks arise from sociotechnical conditions such as communication breakdowns, coordination failures, and organisational culture rather than diagnostic or decision error alone. While simulation can engage these dimensions of care, AI-supported simulation remains limited by heterogeneity and a lack of explicit conceptual structure. This study presents a narrative and conceptual review of the healthcare simulation and AI literature to identify structural barriers to coherent AI reasoning about simulation. Drawing on this synthesis, we introduce Transformative Simulation (TfS) as an intentional framework that can be formalised as an ontology for AI-supported simulation focused on cultural and systems-level change. TfS structures simulation through explicit Simulation-Based Intentions, an aligned design&amp;amp;ndash;delivery&amp;amp;ndash;data&amp;amp;ndash;debrief process, and foundational considerations of purpose, perspective, power, preparation, and possibility. Framed in this way, TfS enables AI systems to interpret simulation artefacts in relation to declared intent, sociotechnical context, and ethical boundaries. We further describe an Intentionality&amp;amp;ndash;Simulation&amp;amp;ndash;Intelligence triad and a continuous learning loop that align human values, simulation structure, and AI reasoning. The findings of this review suggest that an important challenge in applying AI to healthcare simulation may be ontological as well as technical, and that explicit representation of intention and context is necessary to support coherent, context-sensitive, and system-aligned AI reasoning in healthcare.</description>
	<pubDate>2026-06-23</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 203: Transformative Simulation as an Ontology for AI in Health Systems: From Fluent Tools to Coherent Reasoning</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/203">doi: 10.3390/bdcc10070203</a></p>
	<p>Authors:
		Sharon Marie Weldon
		Roger Kneebone
		Fernando Bello
		</p>
	<p>Artificial intelligence (AI) is increasingly applied to healthcare decision-making; however, many persistent patient safety risks arise from sociotechnical conditions such as communication breakdowns, coordination failures, and organisational culture rather than diagnostic or decision error alone. While simulation can engage these dimensions of care, AI-supported simulation remains limited by heterogeneity and a lack of explicit conceptual structure. This study presents a narrative and conceptual review of the healthcare simulation and AI literature to identify structural barriers to coherent AI reasoning about simulation. Drawing on this synthesis, we introduce Transformative Simulation (TfS) as an intentional framework that can be formalised as an ontology for AI-supported simulation focused on cultural and systems-level change. TfS structures simulation through explicit Simulation-Based Intentions, an aligned design&amp;amp;ndash;delivery&amp;amp;ndash;data&amp;amp;ndash;debrief process, and foundational considerations of purpose, perspective, power, preparation, and possibility. Framed in this way, TfS enables AI systems to interpret simulation artefacts in relation to declared intent, sociotechnical context, and ethical boundaries. We further describe an Intentionality&amp;amp;ndash;Simulation&amp;amp;ndash;Intelligence triad and a continuous learning loop that align human values, simulation structure, and AI reasoning. The findings of this review suggest that an important challenge in applying AI to healthcare simulation may be ontological as well as technical, and that explicit representation of intention and context is necessary to support coherent, context-sensitive, and system-aligned AI reasoning in healthcare.</p>
	]]></content:encoded>

	<dc:title>Transformative Simulation as an Ontology for AI in Health Systems: From Fluent Tools to Coherent Reasoning</dc:title>
			<dc:creator>Sharon Marie Weldon</dc:creator>
			<dc:creator>Roger Kneebone</dc:creator>
			<dc:creator>Fernando Bello</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070203</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-06-23</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-06-23</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Review</prism:section>
	<prism:startingPage>203</prism:startingPage>
		<prism:doi>10.3390/bdcc10070203</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/203</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/202">

	<title>BDCC, Vol. 10, Pages 202: Part-of-Speech Context Vectors: Approximating Distributional Meaning of Syntactic Category Symbols</title>
	<link>https://www.mdpi.com/2504-2289/10/7/202</link>
	<description>Words occurring in similar contexts have been observed to have similar meanings. A natural and established method within computational linguistics implements this observation by representing words as vectors with dimensions determined by words that are witnessed in fixed positions in relation to the target word. We generalize this context vector approach to part-of-speech (POS) sequences appropriate to word sequences. As with words, the context of a POS tag (considering the POS tags occurring before and after any target tag) reflects its syntactic constraints and may approximate the &amp;amp;ldquo;meaning&amp;amp;rdquo; of the target tag, from a distributional perspective. We use the 111-million-word British National Corpus (BNC) and the sequence of POS labels lifted from those texts to calculate POS context vectors. We observed significant agreement between the clusters of POS context vectors and the supercategories of corresponding POS tags, and examined potential categorization of the POS categories that emerged from the vector clusters. We also found that though vector measures partially align with the predictions of generativist linguistic theories, the approach suggests a more complex relation between syntactic categories. We conclude that a mutual-information-based approach better approximates the distributional &amp;amp;ldquo;meaning&amp;amp;rdquo; of syntactic categories than the conditional probability distribution of POS symbols.</description>
	<pubDate>2026-06-23</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 202: Part-of-Speech Context Vectors: Approximating Distributional Meaning of Syntactic Category Symbols</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/202">doi: 10.3390/bdcc10070202</a></p>
	<p>Authors:
		Xiaona Ma
		Carl Vogel
		</p>
	<p>Words occurring in similar contexts have been observed to have similar meanings. A natural and established method within computational linguistics implements this observation by representing words as vectors with dimensions determined by words that are witnessed in fixed positions in relation to the target word. We generalize this context vector approach to part-of-speech (POS) sequences appropriate to word sequences. As with words, the context of a POS tag (considering the POS tags occurring before and after any target tag) reflects its syntactic constraints and may approximate the &amp;amp;ldquo;meaning&amp;amp;rdquo; of the target tag, from a distributional perspective. We use the 111-million-word British National Corpus (BNC) and the sequence of POS labels lifted from those texts to calculate POS context vectors. We observed significant agreement between the clusters of POS context vectors and the supercategories of corresponding POS tags, and examined potential categorization of the POS categories that emerged from the vector clusters. We also found that though vector measures partially align with the predictions of generativist linguistic theories, the approach suggests a more complex relation between syntactic categories. We conclude that a mutual-information-based approach better approximates the distributional &amp;amp;ldquo;meaning&amp;amp;rdquo; of syntactic categories than the conditional probability distribution of POS symbols.</p>
	]]></content:encoded>

	<dc:title>Part-of-Speech Context Vectors: Approximating Distributional Meaning of Syntactic Category Symbols</dc:title>
			<dc:creator>Xiaona Ma</dc:creator>
			<dc:creator>Carl Vogel</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070202</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-06-23</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-06-23</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>202</prism:startingPage>
		<prism:doi>10.3390/bdcc10070202</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/202</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/201">

	<title>BDCC, Vol. 10, Pages 201: A Comparative Study of Time Series Clustering Performance with Classification as a Benchmark</title>
	<link>https://www.mdpi.com/2504-2289/10/7/201</link>
	<description>This paper extends a previous classification study by examining clustering methods on the same synthetic datasets and comparing their behavior with the previously obtained classification results. This study investigates the performance of selected time series clustering methods under controlled changes in noise level and class complexity. Six clustering methods representing distance-based, feature-based, and deep learning approaches were evaluated on 82 balanced synthetic datasets. The datasets contained from two to six classes, different levels of additive Gaussian noise, 200 time series per dataset, and 1000 observations per time series. The analysis focused on clustering quality, comparative behavior with classification models, and computational cost in terms of training time and peak memory usage. Clustering quality was assessed mainly using Adjusted Rand Index and V-measure, while accuracy after Hungarian label matching was used as an auxiliary measure for comparison with classification models. The results show that distance-based methods, and particularly TimeSeriesKMedoids, achieved the most robust and consistent clustering performance across the considered settings. Clustering quality decreased with both the number of classes and the noise level, but the effect of noise was clearly stronger. Feature-based and deep learning-based clustering methods were generally more sensitive to noise, while deep models were also associated with substantially higher computational cost. In terms of memory usage, classical clustering methods remained below 50 MiB, whereas deep learning-based clustering methods required substantially more memory. This study further shows that accuracy computed after Hungarian label matching may provide an overly optimistic view of clustering quality. Accuracy after Hungarian label matching is reported only as an auxiliary metric, while the main interpretation of clustering quality is based on structure-sensitive measures such as Adjusted Rand Index and V-measure. Overall, the findings highlight the importance of robust distance-based approaches and of using structure-sensitive evaluation measures when analyzing time series clustering.</description>
	<pubDate>2026-06-23</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 201: A Comparative Study of Time Series Clustering Performance with Classification as a Benchmark</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/201">doi: 10.3390/bdcc10070201</a></p>
	<p>Authors:
		Maria Sadowska
		Krzysztof Gajowniczek
		</p>
	<p>This paper extends a previous classification study by examining clustering methods on the same synthetic datasets and comparing their behavior with the previously obtained classification results. This study investigates the performance of selected time series clustering methods under controlled changes in noise level and class complexity. Six clustering methods representing distance-based, feature-based, and deep learning approaches were evaluated on 82 balanced synthetic datasets. The datasets contained from two to six classes, different levels of additive Gaussian noise, 200 time series per dataset, and 1000 observations per time series. The analysis focused on clustering quality, comparative behavior with classification models, and computational cost in terms of training time and peak memory usage. Clustering quality was assessed mainly using Adjusted Rand Index and V-measure, while accuracy after Hungarian label matching was used as an auxiliary measure for comparison with classification models. The results show that distance-based methods, and particularly TimeSeriesKMedoids, achieved the most robust and consistent clustering performance across the considered settings. Clustering quality decreased with both the number of classes and the noise level, but the effect of noise was clearly stronger. Feature-based and deep learning-based clustering methods were generally more sensitive to noise, while deep models were also associated with substantially higher computational cost. In terms of memory usage, classical clustering methods remained below 50 MiB, whereas deep learning-based clustering methods required substantially more memory. This study further shows that accuracy computed after Hungarian label matching may provide an overly optimistic view of clustering quality. Accuracy after Hungarian label matching is reported only as an auxiliary metric, while the main interpretation of clustering quality is based on structure-sensitive measures such as Adjusted Rand Index and V-measure. Overall, the findings highlight the importance of robust distance-based approaches and of using structure-sensitive evaluation measures when analyzing time series clustering.</p>
	]]></content:encoded>

	<dc:title>A Comparative Study of Time Series Clustering Performance with Classification as a Benchmark</dc:title>
			<dc:creator>Maria Sadowska</dc:creator>
			<dc:creator>Krzysztof Gajowniczek</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070201</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-06-23</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-06-23</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>201</prism:startingPage>
		<prism:doi>10.3390/bdcc10070201</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/201</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/200">

	<title>BDCC, Vol. 10, Pages 200: Bio-Inspired Spiking Recurrent Networks with Evolutionary Optimization for Non-Stationary Cryptocurrency Forecasting</title>
	<link>https://www.mdpi.com/2504-2289/10/7/200</link>
	<description>Forecasting cryptocurrency prices remains difficult because market dynamics are highly volatile, non-stationary, and regime-dependent. This study investigates whether combining a spiking-inspired recurrent architecture with the Grey Wolf Optimizer (GWO) can improve one-step-ahead Bitcoin forecasting within a controlled model family. We compare four configurations, LSTM, SLSTM, GWO-LSTM, and GWO-SLSTM, on 4039 daily BTC&amp;amp;ndash;USD closing prices from 17 September 2014 to 9 October 2025 using Min&amp;amp;ndash;Max normalization, strict chronological splitting, windowed regime-based robustness analysis across three distinct market regimes, and repeated-run testing. The proposed SLSTM replaces the conventional hidden-state recurrence with leaky integrate-and-fire-inspired synaptic, membrane, and adaptive-threshold dynamics, functioning as a spiking-inspired recurrent model with thresholded event gating (reset = `none&amp;amp;rsquo;, learnable threshold). On the primary hold-out split, GWO-SLSTM achieved a test RMSE of 1840.97 and a test MAPE of 1.76%, compared with 2217.24 and 2.46% for GWO-LSTM, 3501.48 and 3.86% for SLSTM, and 4030.10 and 4.40% for LSTM. Both GWO-optimized models exhibited substantial improvements over their non-optimized counterparts, while the SLSTM baseline also outperformed the plain LSTM, indicating gains from both spiking-inspired recurrence and evolutionary hyperparameter optimization. Both optimized models exhibited near-zero bias (PBIAS 0.11% for GWO-LSTM and 0.36% for GWO-SLSTM). Within the present implementation, GWO-SLSTM also trained faster than GWO-LSTM (39.71 s vs. 137.28 s), although this runtime difference should be interpreted as setup-specific because the model families were implemented in different frameworks and stopped after different numbers of epochs. Overall, within the expanded univariate BTC&amp;amp;ndash;USD setting, the results support GWO-SLSTM as a strong within-family candidate for one-step-ahead forecasting under non-stationary conditions.</description>
	<pubDate>2026-06-23</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 200: Bio-Inspired Spiking Recurrent Networks with Evolutionary Optimization for Non-Stationary Cryptocurrency Forecasting</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/200">doi: 10.3390/bdcc10070200</a></p>
	<p>Authors:
		Francis Noah Walugembe
		Maciej Wielgosz
		Matej Mertik
		Matjaž Gams
		</p>
	<p>Forecasting cryptocurrency prices remains difficult because market dynamics are highly volatile, non-stationary, and regime-dependent. This study investigates whether combining a spiking-inspired recurrent architecture with the Grey Wolf Optimizer (GWO) can improve one-step-ahead Bitcoin forecasting within a controlled model family. We compare four configurations, LSTM, SLSTM, GWO-LSTM, and GWO-SLSTM, on 4039 daily BTC&amp;amp;ndash;USD closing prices from 17 September 2014 to 9 October 2025 using Min&amp;amp;ndash;Max normalization, strict chronological splitting, windowed regime-based robustness analysis across three distinct market regimes, and repeated-run testing. The proposed SLSTM replaces the conventional hidden-state recurrence with leaky integrate-and-fire-inspired synaptic, membrane, and adaptive-threshold dynamics, functioning as a spiking-inspired recurrent model with thresholded event gating (reset = `none&amp;amp;rsquo;, learnable threshold). On the primary hold-out split, GWO-SLSTM achieved a test RMSE of 1840.97 and a test MAPE of 1.76%, compared with 2217.24 and 2.46% for GWO-LSTM, 3501.48 and 3.86% for SLSTM, and 4030.10 and 4.40% for LSTM. Both GWO-optimized models exhibited substantial improvements over their non-optimized counterparts, while the SLSTM baseline also outperformed the plain LSTM, indicating gains from both spiking-inspired recurrence and evolutionary hyperparameter optimization. Both optimized models exhibited near-zero bias (PBIAS 0.11% for GWO-LSTM and 0.36% for GWO-SLSTM). Within the present implementation, GWO-SLSTM also trained faster than GWO-LSTM (39.71 s vs. 137.28 s), although this runtime difference should be interpreted as setup-specific because the model families were implemented in different frameworks and stopped after different numbers of epochs. Overall, within the expanded univariate BTC&amp;amp;ndash;USD setting, the results support GWO-SLSTM as a strong within-family candidate for one-step-ahead forecasting under non-stationary conditions.</p>
	]]></content:encoded>

	<dc:title>Bio-Inspired Spiking Recurrent Networks with Evolutionary Optimization for Non-Stationary Cryptocurrency Forecasting</dc:title>
			<dc:creator>Francis Noah Walugembe</dc:creator>
			<dc:creator>Maciej Wielgosz</dc:creator>
			<dc:creator>Matej Mertik</dc:creator>
			<dc:creator>Matjaž Gams</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070200</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-06-23</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-06-23</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>200</prism:startingPage>
		<prism:doi>10.3390/bdcc10070200</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/200</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/7/199">

	<title>BDCC, Vol. 10, Pages 199: Semantic Analysis of Technical Documentation: Systematic Review, Formal Task Definition, and Transformer-Based NER Implementation</title>
	<link>https://www.mdpi.com/2504-2289/10/7/199</link>
	<description>The increasing complexity and volume of technical documentation, including requirements specifications, patents, and engineering reports, create significant challenges for manual analysis and knowledge extraction. This paper includes a systematic review of methods for semantic content analysis of technical documents, with a particular focus on Natural Language Processing (NLP) techniques and Transformer-based models. The study formalizes the task of structured information extraction and provides a mathematical description of Named Entity Recognition (NER) as a core subtask. A practical case study demonstrates an end-to-end NER pipeline for Russian-language technical requirements, leveraging ruRoberta-large via spaCy-transformers. The results highlight both the potential and limitations of current approaches, emphasizing the critical role of annotation consistency and document format normalization. This work contributes to the development of intelligent systems for engineering documentation analysis and outlines key directions for future research.</description>
	<pubDate>2026-06-23</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 199: Semantic Analysis of Technical Documentation: Systematic Review, Formal Task Definition, and Transformer-Based NER Implementation</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/7/199">doi: 10.3390/bdcc10070199</a></p>
	<p>Authors:
		Alexander Echin
		Alla G. Kravets
		Elena Safonova
		Dmitry A. Skorobogatchenko
		Danila Karasev
		</p>
	<p>The increasing complexity and volume of technical documentation, including requirements specifications, patents, and engineering reports, create significant challenges for manual analysis and knowledge extraction. This paper includes a systematic review of methods for semantic content analysis of technical documents, with a particular focus on Natural Language Processing (NLP) techniques and Transformer-based models. The study formalizes the task of structured information extraction and provides a mathematical description of Named Entity Recognition (NER) as a core subtask. A practical case study demonstrates an end-to-end NER pipeline for Russian-language technical requirements, leveraging ruRoberta-large via spaCy-transformers. The results highlight both the potential and limitations of current approaches, emphasizing the critical role of annotation consistency and document format normalization. This work contributes to the development of intelligent systems for engineering documentation analysis and outlines key directions for future research.</p>
	]]></content:encoded>

	<dc:title>Semantic Analysis of Technical Documentation: Systematic Review, Formal Task Definition, and Transformer-Based NER Implementation</dc:title>
			<dc:creator>Alexander Echin</dc:creator>
			<dc:creator>Alla G. Kravets</dc:creator>
			<dc:creator>Elena Safonova</dc:creator>
			<dc:creator>Dmitry A. Skorobogatchenko</dc:creator>
			<dc:creator>Danila Karasev</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10070199</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-06-23</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-06-23</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>7</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>199</prism:startingPage>
		<prism:doi>10.3390/bdcc10070199</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/7/199</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/6/198">

	<title>BDCC, Vol. 10, Pages 198: Improving Entity Understanding for Vision-Language Pre-Training via Active Learning</title>
	<link>https://www.mdpi.com/2504-2289/10/6/198</link>
	<description>Although many researchers use pre-trained models to better solve downstream tasks, further exploration of more effective pre-training methods remains necessary, especially for multi-modal pre-training where high-quality training data is more difficult to obtain. This work aims to improve the knowledge-learning performance in multi-modal pre-training. Some researchers focus on injecting entity knowledge into language pre-trained models based on masked entity model (MEM) training, which masks entities randomly and lets the model recover. These methods cannot guarantee good performance due to the lack of consideration of which entities are more valuable for learning. Moreover, in multi-modal training data, some entities may be unrelated to visual content. In this work, for the vision-language pre-trained model, we propose a Masked Entity Model pre-training method based on Active learning (ActiveMEM). It is designed to actively mask important and informative entities&amp;amp;mdash;those that are both informative and uncertain&amp;amp;mdash;for the model to recover, thereby encouraging it to extract more valuable knowledge from the data. The proposed method is evaluated using three pre-training datasets and four downstream datasets, and the experimental results demonstrate the effectiveness of our method.</description>
	<pubDate>2026-06-22</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 198: Improving Entity Understanding for Vision-Language Pre-Training via Active Learning</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/6/198">doi: 10.3390/bdcc10060198</a></p>
	<p>Authors:
		Qunbo Wang
		Sen Zhang
		Boxuan Shao
		Xize Guo
		Jiayong An
		Chao Fan
		Yuanjun Jing
		Junxian Li
		Wenjun Wu
		</p>
	<p>Although many researchers use pre-trained models to better solve downstream tasks, further exploration of more effective pre-training methods remains necessary, especially for multi-modal pre-training where high-quality training data is more difficult to obtain. This work aims to improve the knowledge-learning performance in multi-modal pre-training. Some researchers focus on injecting entity knowledge into language pre-trained models based on masked entity model (MEM) training, which masks entities randomly and lets the model recover. These methods cannot guarantee good performance due to the lack of consideration of which entities are more valuable for learning. Moreover, in multi-modal training data, some entities may be unrelated to visual content. In this work, for the vision-language pre-trained model, we propose a Masked Entity Model pre-training method based on Active learning (ActiveMEM). It is designed to actively mask important and informative entities&amp;amp;mdash;those that are both informative and uncertain&amp;amp;mdash;for the model to recover, thereby encouraging it to extract more valuable knowledge from the data. The proposed method is evaluated using three pre-training datasets and four downstream datasets, and the experimental results demonstrate the effectiveness of our method.</p>
	]]></content:encoded>

	<dc:title>Improving Entity Understanding for Vision-Language Pre-Training via Active Learning</dc:title>
			<dc:creator>Qunbo Wang</dc:creator>
			<dc:creator>Sen Zhang</dc:creator>
			<dc:creator>Boxuan Shao</dc:creator>
			<dc:creator>Xize Guo</dc:creator>
			<dc:creator>Jiayong An</dc:creator>
			<dc:creator>Chao Fan</dc:creator>
			<dc:creator>Yuanjun Jing</dc:creator>
			<dc:creator>Junxian Li</dc:creator>
			<dc:creator>Wenjun Wu</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10060198</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-06-22</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-06-22</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>6</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>198</prism:startingPage>
		<prism:doi>10.3390/bdcc10060198</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/6/198</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/6/197">

	<title>BDCC, Vol. 10, Pages 197: Green Cryptos or Echo Chambers? Analyzing Community Discourse on Blockchain Environmental Impacts</title>
	<link>https://www.mdpi.com/2504-2289/10/6/197</link>
	<description>As the environmental sustainability of blockchain technology becomes a focal point of public and academic debate, understanding how technically engaged communities frame this issue is increasingly important. This study examines 3000 long-form comments from a highly active sustainability-focused Bitcointalk thread to analyze sentiment patterns, recurring arguments, and the linguistic cues associated with community responses to environmental criticism. Using Natural Language Processing (NLP) methods, we apply Valence Aware Dictionary and sEntiment Reasoner (VADER) sentiment analysis to classify the discourse, n-gram extraction to identify dominant thematic expressions, and a Random Forest model combined with SHapley Additive exPlanations (SHAP) to interpret the lexical features most strongly associated with sentiment polarity. The results show a strongly positive and internally consistent discourse structure: 87.63% of comments are classified as positive, while negative and neutral comments are comparatively rare. The dominant themes emphasize energy consumption as a necessary trade-off for network security, while external criticism is frequently reframed or rejected. Explanatory modeling further indicates that negative sentiment is primarily driven by terms associated with climate risk, damage, and reputational concerns when users respond to criticism. Rather than claiming to capture the cryptocurrency ecosystem as a whole, this study presents a localized case study of one Bitcointalk mega-thread and describes it as a highly homogeneous narrative space shaped by recurrent rebuttal and rhetorical reinforcement. The findings offer a focused contribution to understanding how insider communities construct sustainability narratives around blockchain energy use, while also highlighting the need for broader comparative and network-structural research in future work.</description>
	<pubDate>2026-06-21</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 197: Green Cryptos or Echo Chambers? Analyzing Community Discourse on Blockchain Environmental Impacts</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/6/197">doi: 10.3390/bdcc10060197</a></p>
	<p>Authors:
		Parisa Bouzari
		Maria Fekete-Farkas
		Zsigmond Gábor Szalay
		</p>
	<p>As the environmental sustainability of blockchain technology becomes a focal point of public and academic debate, understanding how technically engaged communities frame this issue is increasingly important. This study examines 3000 long-form comments from a highly active sustainability-focused Bitcointalk thread to analyze sentiment patterns, recurring arguments, and the linguistic cues associated with community responses to environmental criticism. Using Natural Language Processing (NLP) methods, we apply Valence Aware Dictionary and sEntiment Reasoner (VADER) sentiment analysis to classify the discourse, n-gram extraction to identify dominant thematic expressions, and a Random Forest model combined with SHapley Additive exPlanations (SHAP) to interpret the lexical features most strongly associated with sentiment polarity. The results show a strongly positive and internally consistent discourse structure: 87.63% of comments are classified as positive, while negative and neutral comments are comparatively rare. The dominant themes emphasize energy consumption as a necessary trade-off for network security, while external criticism is frequently reframed or rejected. Explanatory modeling further indicates that negative sentiment is primarily driven by terms associated with climate risk, damage, and reputational concerns when users respond to criticism. Rather than claiming to capture the cryptocurrency ecosystem as a whole, this study presents a localized case study of one Bitcointalk mega-thread and describes it as a highly homogeneous narrative space shaped by recurrent rebuttal and rhetorical reinforcement. The findings offer a focused contribution to understanding how insider communities construct sustainability narratives around blockchain energy use, while also highlighting the need for broader comparative and network-structural research in future work.</p>
	]]></content:encoded>

	<dc:title>Green Cryptos or Echo Chambers? Analyzing Community Discourse on Blockchain Environmental Impacts</dc:title>
			<dc:creator>Parisa Bouzari</dc:creator>
			<dc:creator>Maria Fekete-Farkas</dc:creator>
			<dc:creator>Zsigmond Gábor Szalay</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10060197</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-06-21</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-06-21</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>6</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>197</prism:startingPage>
		<prism:doi>10.3390/bdcc10060197</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/6/197</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
        <item rdf:about="https://www.mdpi.com/2504-2289/10/6/195">

	<title>BDCC, Vol. 10, Pages 195: HiCoPro: A Graph-Conditioned Structured Inference Framework for Hierarchical Dialogue Semantic Path Prediction</title>
	<link>https://www.mdpi.com/2504-2289/10/6/195</link>
	<description>Most existing dialogue understanding methods rely on flat classification paradigms, failing to capture hierarchical semantic structures and cross-level dependencies. To address this limitation, we reformulate dialogue understanding as a hierarchical semantic path inference problem, where prediction is performed over a constrained path space rather than independent label spaces. We propose HiCoPro, a graph-conditioned structured inference framework for modeling multi-level dialogue semantics. The framework consists of the following: (i) a Graph-Conditioned Label Space (GCLS) that encodes hierarchical dependencies into label embeddings via graph propagation; (ii) a compatibility-based logit fusion mechanism that jointly scores semantic relevance and structural consistency; and (iii) a constraint-aware decoding strategy that enforces hard parent&amp;amp;ndash;child dependencies during inference. By integrating semantic representations with graph-conditioned label structures via a bilinear compatibility function and learnable logit-level fusion, the model jointly captures semantic relevance and structural consistency. To support this task, we construct PrefDial, a general-domain hierarchical dialogue dataset with systematic three-level annotations, serving as a benchmark for structured dialogue understanding. Experimental results demonstrate that HiCoPro achieves superior Macro F1, Exact Match, and Hierarchical Consistency on PrefDial, while remaining competitive on multiple public benchmarks. Further analysis highlights the effectiveness of graph-conditioned modeling in balancing semantic discrimination, hierarchical consistency, and robustness.</description>
	<pubDate>2026-06-21</pubDate>

	<content:encoded><![CDATA[
	<p><b>BDCC, Vol. 10, Pages 195: HiCoPro: A Graph-Conditioned Structured Inference Framework for Hierarchical Dialogue Semantic Path Prediction</b></p>
	<p>Big Data and Cognitive Computing <a href="https://www.mdpi.com/2504-2289/10/6/195">doi: 10.3390/bdcc10060195</a></p>
	<p>Authors:
		Yulin Yang
		Jinglan Zhang
		Xinyi Chen
		Shijie Fu
		Bin Ai
		</p>
	<p>Most existing dialogue understanding methods rely on flat classification paradigms, failing to capture hierarchical semantic structures and cross-level dependencies. To address this limitation, we reformulate dialogue understanding as a hierarchical semantic path inference problem, where prediction is performed over a constrained path space rather than independent label spaces. We propose HiCoPro, a graph-conditioned structured inference framework for modeling multi-level dialogue semantics. The framework consists of the following: (i) a Graph-Conditioned Label Space (GCLS) that encodes hierarchical dependencies into label embeddings via graph propagation; (ii) a compatibility-based logit fusion mechanism that jointly scores semantic relevance and structural consistency; and (iii) a constraint-aware decoding strategy that enforces hard parent&amp;amp;ndash;child dependencies during inference. By integrating semantic representations with graph-conditioned label structures via a bilinear compatibility function and learnable logit-level fusion, the model jointly captures semantic relevance and structural consistency. To support this task, we construct PrefDial, a general-domain hierarchical dialogue dataset with systematic three-level annotations, serving as a benchmark for structured dialogue understanding. Experimental results demonstrate that HiCoPro achieves superior Macro F1, Exact Match, and Hierarchical Consistency on PrefDial, while remaining competitive on multiple public benchmarks. Further analysis highlights the effectiveness of graph-conditioned modeling in balancing semantic discrimination, hierarchical consistency, and robustness.</p>
	]]></content:encoded>

	<dc:title>HiCoPro: A Graph-Conditioned Structured Inference Framework for Hierarchical Dialogue Semantic Path Prediction</dc:title>
			<dc:creator>Yulin Yang</dc:creator>
			<dc:creator>Jinglan Zhang</dc:creator>
			<dc:creator>Xinyi Chen</dc:creator>
			<dc:creator>Shijie Fu</dc:creator>
			<dc:creator>Bin Ai</dc:creator>
		<dc:identifier>doi: 10.3390/bdcc10060195</dc:identifier>
	<dc:source>Big Data and Cognitive Computing</dc:source>
	<dc:date>2026-06-21</dc:date>

	<prism:publicationName>Big Data and Cognitive Computing</prism:publicationName>
	<prism:publicationDate>2026-06-21</prism:publicationDate>
	<prism:volume>10</prism:volume>
	<prism:number>6</prism:number>
	<prism:section>Article</prism:section>
	<prism:startingPage>195</prism:startingPage>
		<prism:doi>10.3390/bdcc10060195</prism:doi>
	<prism:url>https://www.mdpi.com/2504-2289/10/6/195</prism:url>
	
	<cc:license rdf:resource="CC BY 4.0"/>
</item>
    
<cc:License rdf:about="https://creativecommons.org/licenses/by/4.0/">
	<cc:permits rdf:resource="https://creativecommons.org/ns#Reproduction" />
	<cc:permits rdf:resource="https://creativecommons.org/ns#Distribution" />
	<cc:permits rdf:resource="https://creativecommons.org/ns#DerivativeWorks" />
</cc:License>

</rdf:RDF>
