From 5aa995497e828affbc72bd488b383539bd0a5c76 Mon Sep 17 00:00:00 2001 From: "F.Tibor" Date: Sun, 5 Jul 2026 14:07:38 +0200 Subject: [PATCH 1/5] Merge metadata files --- src/BUILD | 7 +++ src/metadata_merge.py | 124 ++++++++++++++++++++++++++++++++++++++++++ src/per_file.bzl | 24 ++++++++ 3 files changed, 155 insertions(+) create mode 100644 src/metadata_merge.py diff --git a/src/BUILD b/src/BUILD index a43a5b3f..bc99daf8 100644 --- a/src/BUILD +++ b/src/BUILD @@ -20,6 +20,13 @@ py_binary( visibility = ["//visibility:public"], ) +py_binary( + name = "metadata_merge", + srcs = ["metadata_merge.py"], + # Bazel 6 won't see this otherwise + visibility = ["//visibility:public"], +) + py_binary( name = "codechecker_script", srcs = ["codechecker_script.py"], diff --git a/src/metadata_merge.py b/src/metadata_merge.py new file mode 100644 index 00000000..b034ea28 --- /dev/null +++ b/src/metadata_merge.py @@ -0,0 +1,124 @@ +# Copyright 2023 Ericsson AB +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +""" +Merges the metadata information of multiple CodeChecker analysis. +""" + +import json +import os +import sys +from typing import List, Dict, Any + + +def merge_two_json(json1, json2): + """ + Merges the data of two json files. + + If both json files are empty, returns an empty json. + """ + if json1 == {}: + return json2 + if json2 == {}: + return json1 + # Happens when analysis of all files was skipped + if json1 == {} and json2 == {}: + return {} + # Fail if the plist file version is different + assert json1["version"] == json2["version"] + json1_root = json1["tools"][0] + json2_root = json2["tools"][0] + # Command, working directory nad output directory may be different + # from metadata to metadata, due to remote workers. + # Currently we choose the first of these values. + # We expect the following fields to be the same in all metadata files. + assert json1_root["name"] == json2_root["name"] + # same CodeChecker version + assert json1_root["version"] == json2_root["version"] + # We assume that the list of enabled checkers haven't changed between runs. + # We append info from json2 to json1 from here on out + json1_root["action_num"] += json2_root["action_num"] + json1_root["result_source_files"].update(json2_root["result_source_files"]) + json1_root["skipped"] = json1_root["skipped"] + json2_root["skipped"] + # Merge time; we assume here both json files describe jobs in + # the same analysis invocation, implying that the analysis start + # time is the lowest timestamp, and the end is the highest. + # Note: caching will break this assumption + json1_root["timestamps"]["begin"] = min( + float(json1_root["timestamps"]["begin"]), + float(json2_root["timestamps"]["begin"]), + ) + json1_root["timestamps"]["end"] = max( + float(json1_root["timestamps"]["end"]), + float(json2_root["timestamps"]["end"]), + ) + # Merge analyzers + for key, _ in json2_root["analyzers"].items(): + json1_stat = json1_root["analyzers"][key]["analyzer_statistics"] + json2_stat = json2_root["analyzers"][key]["analyzer_statistics"] + json1_stat["failed"] = json1_stat["failed"] + json2_stat["failed"] + json1_stat["failed_sources"].extend(json2_stat["failed_sources"]) + json1_stat["successful"] = ( + json1_stat["successful"] + json2_stat["successful"] + ) + json1_stat["successful_sources"].extend( + json2_stat["successful_sources"] + ) + return json1 + + +def merge_json_files(file_paths: List[str]) -> Dict[str, Any]: + """ + Merges a list of metadata.json files, using merge_two_json. + + Returns the merged contents of the json files. + """ + merged_data = {} + for file_path in file_paths: + if not os.path.exists(file_path): + print( + f"Error: File not found at '{file_path}'. Skipping.", + file=sys.stderr, + ) + continue + + try: + with open(file_path, "r", encoding="utf-8") as f: + data = json.load(f) + merged_data = merge_two_json(merged_data, data) + except json.JSONDecodeError: + print( + f"Error: Could not decode JSON from '{file_path}'. Skipping.", + file=sys.stderr, + ) + + return merged_data + +def main(): + """ + Main function of metadata merge + """ + output_file = sys.argv[1] + input_files = sys.argv[2:] + + merged_data = merge_json_files(input_files) + if merged_data: + with open(output_file, "w", encoding="utf-8") as f: + json.dump(merged_data, f, indent=4) + else: + print("\nNo data was merged. Output file will not be created.") + + +if __name__ == "__main__": + main() diff --git a/src/per_file.bzl b/src/per_file.bzl index d33b85dd..d184a0d5 100644 --- a/src/per_file.bzl +++ b/src/per_file.bzl @@ -180,6 +180,24 @@ def _collect_all_sources_and_headers(ctx): all_files += headers return all_files +def _merge_metadata(ctx, all_files): + """ + Merges metadata files of individual CodeChecker runs into 1 + + Returns the metadata file objects + """ + metadata = [file for file in all_files if file.path.endswith("metadata.json")] + metadata_json = ctx.actions.declare_file(ctx.attr.name + "/data/metadata.json") + ctx.actions.run( + inputs = metadata, + outputs = [metadata_json], + executable = ctx.executable._metadata_merge, + arguments = [metadata_json.path] + [file.path for file in metadata], + mnemonic = "Metadata", + progress_message = "Merging metadata.json", + ) + return metadata_json + def _per_file_impl(ctx): info = ctx.toolchains["//:toolchain_type"].codecheckerinfo compile_commands = None @@ -229,6 +247,7 @@ def _per_file_impl(ctx): sources_and_headers, ) all_files += outputs + all_files.append(_merge_metadata(ctx, all_files)) ctx.actions.write( output = ctx.outputs.test_script, is_executable = True, @@ -296,6 +315,11 @@ per_file_test = rule( "When set, tools from this target are used instead of " + "Bazel's toolchain resolution.", ), + "_metadata_merge": attr.label( + default = ":metadata_merge", + executable = True, + cfg = "exec", + ), "_per_file_script": attr.label( executable = True, cfg = "exec", From 322ccea8ba1b8fae1f051410f8d59d67ee7fbaa9 Mon Sep 17 00:00:00 2001 From: "F.Tibor" Date: Sun, 5 Jul 2026 14:26:53 +0200 Subject: [PATCH 2/5] Remove individual metadata files from defaultinfo We absolutely do not need the individual metadata files in the data folder, the merged one is enough --- src/per_file.bzl | 16 +++++++++------- 1 file changed, 9 insertions(+), 7 deletions(-) diff --git a/src/per_file.bzl b/src/per_file.bzl index d184a0d5..e2d4f190 100644 --- a/src/per_file.bzl +++ b/src/per_file.bzl @@ -90,7 +90,6 @@ def _run_code_checker( clang_tidy_plist, clangsa_plist, codechecker_log, - codechecker_metadata, ] analyzer_output_paths = "clangsa," + clangsa_plist.path + \ @@ -105,7 +104,9 @@ def _run_code_checker( # TODO: use env for environment variables, instead of passing it as argument ctx.actions.run( inputs = inputs, - outputs = outputs, + # We do not want all individual metadata files + # cluttering the data folder + outputs = outputs + [codechecker_metadata], executable = per_file_script, tools = [ info.runfiles, @@ -137,7 +138,7 @@ def _run_code_checker( mnemonic = "CodeChecker", progress_message = "CodeChecker analyze {}".format(src.short_path), ) - return outputs + return outputs, codechecker_metadata def check_valid_file_type(src): """ @@ -180,13 +181,12 @@ def _collect_all_sources_and_headers(ctx): all_files += headers return all_files -def _merge_metadata(ctx, all_files): +def _merge_metadata(ctx, metadata): """ Merges metadata files of individual CodeChecker runs into 1 Returns the metadata file objects """ - metadata = [file for file in all_files if file.path.endswith("metadata.json")] metadata_json = ctx.actions.declare_file(ctx.attr.name + "/data/metadata.json") ctx.actions.run( inputs = metadata, @@ -210,6 +210,7 @@ def _per_file_impl(ctx): fail("Seems compile_commands.json file is incorrect!") sources_and_headers = _collect_all_sources_and_headers(ctx) options = ctx.attr.default_options + ctx.attr.options + all_metadata = [] config_file, env_vars = get_config_file(ctx) all_files = [compile_commands, config_file] @@ -231,7 +232,7 @@ def _per_file_impl(ctx): if not check_valid_file_type(src): continue args = target[SourceFilesInfo].compilation_db.to_list() - outputs = _run_code_checker( + outputs, metadata = _run_code_checker( ctx, per_file_script, src, @@ -247,7 +248,8 @@ def _per_file_impl(ctx): sources_and_headers, ) all_files += outputs - all_files.append(_merge_metadata(ctx, all_files)) + all_metadata.append(metadata) + all_files.append(_merge_metadata(ctx, all_metadata)) ctx.actions.write( output = ctx.outputs.test_script, is_executable = True, From e02aa1c52bf18a7d4f78d0c36b3c07d7e4514c5d Mon Sep 17 00:00:00 2001 From: "F.Tibor" Date: Tue, 18 Aug 2026 13:30:31 +0200 Subject: [PATCH 3/5] Extend metadata merging according to the metadata docs --- src/metadata_merge.py | 148 +++++++++++++++++++++++++++++++----------- 1 file changed, 110 insertions(+), 38 deletions(-) diff --git a/src/metadata_merge.py b/src/metadata_merge.py index b034ea28..02f97325 100644 --- a/src/metadata_merge.py +++ b/src/metadata_merge.py @@ -22,11 +22,80 @@ from typing import List, Dict, Any +def _merge_analyzer_statistics(stat1, stat2): + """ + Merges two analyzer_statistics dicts into stat1. + + Sums failed/successful counts, extends source lists, + and preserves the version field. + """ + stat1["failed"] = stat1["failed"] + stat2["failed"] + stat1["failed_sources"].extend(stat2["failed_sources"]) + stat1["successful"] = stat1["successful"] + stat2["successful"] + stat1["successful_sources"].extend(stat2["successful_sources"]) + + +def _merge_checkers(checkers1, checkers2): + """ + Merges two checker dicts. + + A checker is enabled (True) in the result if it is enabled + in either input. This handles the case where different + per-file runs may report slightly different checker sets. + """ + for checker, enabled in checkers2.items(): + if checker not in checkers1: + checkers1[checker] = enabled + else: + # If enabled in either, mark as enabled + checkers1[checker] = checkers1[checker] or enabled + + +def _merge_analyzers(analyzers1, analyzers2): + """ + Merges analyzer sections from two metadata files. + + Handles the case where json2 has analyzers not present + in json1 by initializing them from json2. + """ + for analyzer_name, analyzer_data in analyzers2.items(): + if analyzer_name not in analyzers1: + # Analyzer exists in json2 but not json1; adopt it + analyzers1[analyzer_name] = analyzer_data + continue + + # Merge checkers + if "checkers" in analyzer_data: + if "checkers" not in analyzers1[analyzer_name]: + analyzers1[analyzer_name]["checkers"] = {} + _merge_checkers( + analyzers1[analyzer_name]["checkers"], + analyzer_data["checkers"], + ) + + # Merge analyzer_statistics + if "analyzer_statistics" in analyzer_data: + if "analyzer_statistics" not in analyzers1[analyzer_name]: + analyzers1[analyzer_name]["analyzer_statistics"] = ( + analyzer_data["analyzer_statistics"] + ) + else: + _merge_analyzer_statistics( + analyzers1[analyzer_name]["analyzer_statistics"], + analyzer_data["analyzer_statistics"], + ) + + def merge_two_json(json1, json2): """ - Merges the data of two json files. + Merges the data of two metadata json files. If both json files are empty, returns an empty json. + Handles all fields defined in the CodeChecker metadata spec: + - version, name, action_num, command, working_directory, + output_path, result_source_files, analyzers (with checkers + and analyzer_statistics including version), skipped, + timestamps. """ if json1 == {}: return json2 @@ -35,46 +104,49 @@ def merge_two_json(json1, json2): # Happens when analysis of all files was skipped if json1 == {} and json2 == {}: return {} - # Fail if the plist file version is different - assert json1["version"] == json2["version"] - json1_root = json1["tools"][0] - json2_root = json2["tools"][0] - # Command, working directory nad output directory may be different - # from metadata to metadata, due to remote workers. - # Currently we choose the first of these values. - # We expect the following fields to be the same in all metadata files. - assert json1_root["name"] == json2_root["name"] - # same CodeChecker version - assert json1_root["version"] == json2_root["version"] - # We assume that the list of enabled checkers haven't changed between runs. - # We append info from json2 to json1 from here on out - json1_root["action_num"] += json2_root["action_num"] - json1_root["result_source_files"].update(json2_root["result_source_files"]) - json1_root["skipped"] = json1_root["skipped"] + json2_root["skipped"] - # Merge time; we assume here both json files describe jobs in - # the same analysis invocation, implying that the analysis start - # time is the lowest timestamp, and the end is the highest. + # Fail if the metadata format version is not 2 + assert json1["version"] == 2 + assert json2["version"] == 2 + json1_tools = json1["tools"][0] + json2_tools = json2["tools"][0] + # We expect the following fields to be the same in all + # metadata files from the same analysis invocation. + assert json1_tools["name"] == json2_tools["name"] + # Same CodeChecker version + assert json1_tools["version"] == json2_tools["version"] + # command, working_directory and output_path may differ + # between per-file runs (e.g. remote workers). We keep + # json1's values as the canonical ones. + + # Sum action counts and skipped files + json1_tools["action_num"] += json2_tools["action_num"] + json1_tools["skipped"] = json1_tools["skipped"] + json2_tools["skipped"] + + # Merge result_source_files mapping + json1_tools["result_source_files"].update( + json2_tools["result_source_files"] + ) + + # Merge timestamps; we assume both json files describe jobs + # in the same analysis invocation, implying that the analysis + # start time is the lowest timestamp, and the end is the + # highest. # Note: caching will break this assumption - json1_root["timestamps"]["begin"] = min( - float(json1_root["timestamps"]["begin"]), - float(json2_root["timestamps"]["begin"]), + # Users may see months, or even years long difference in timestamps + json1_tools["timestamps"]["begin"] = min( + float(json1_tools["timestamps"]["begin"]), + float(json2_tools["timestamps"]["begin"]), ) - json1_root["timestamps"]["end"] = max( - float(json1_root["timestamps"]["end"]), - float(json2_root["timestamps"]["end"]), + json1_tools["timestamps"]["end"] = max( + float(json1_tools["timestamps"]["end"]), + float(json2_tools["timestamps"]["end"]), ) - # Merge analyzers - for key, _ in json2_root["analyzers"].items(): - json1_stat = json1_root["analyzers"][key]["analyzer_statistics"] - json2_stat = json2_root["analyzers"][key]["analyzer_statistics"] - json1_stat["failed"] = json1_stat["failed"] + json2_stat["failed"] - json1_stat["failed_sources"].extend(json2_stat["failed_sources"]) - json1_stat["successful"] = ( - json1_stat["successful"] + json2_stat["successful"] - ) - json1_stat["successful_sources"].extend( - json2_stat["successful_sources"] - ) + + # Merge analyzers (checkers + analyzer_statistics) + _merge_analyzers( + json1_tools["analyzers"], json2_tools["analyzers"] + ) + return json1 From 781acdf2347b1c134168405a264ca1961627aae7 Mon Sep 17 00:00:00 2001 From: "F.Tibor" Date: Wed, 19 Aug 2026 10:43:22 +0200 Subject: [PATCH 4/5] Fix tests --- test/unit/metadata/BUILD | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/test/unit/metadata/BUILD b/test/unit/metadata/BUILD index 8c22096f..52fea0ac 100644 --- a/test/unit/metadata/BUILD +++ b/test/unit/metadata/BUILD @@ -61,10 +61,9 @@ unit_test( unit_test( name = "per_file_metadata_test", - contains = [r"\"action_num\": 1,"], + contains = [r"\"action_num\": 2,"], data = [":per_file_multiple_source"], files = [ - "test/unit/metadata/per_file_multiple_source/data/test-unit-metadata-main.cc_metadata.json", - "test/unit/metadata/per_file_multiple_source/data/test-unit-metadata-other.cc_metadata.json", + "test/unit/metadata/per_file_multiple_source/data/metadata.json", ], ) From 3c1dce721e0240458e75ddeed81a3d1065535a7b Mon Sep 17 00:00:00 2001 From: "F.Tibor" Date: Wed, 19 Aug 2026 10:47:29 +0200 Subject: [PATCH 5/5] Add comment to point to structure define of metadata.json --- src/metadata_merge.py | 3 +++ 1 file changed, 3 insertions(+) diff --git a/src/metadata_merge.py b/src/metadata_merge.py index 02f97325..8647fe31 100644 --- a/src/metadata_merge.py +++ b/src/metadata_merge.py @@ -22,6 +22,9 @@ from typing import List, Dict, Any +# Structure of metadata files is defined here: +# https://github.com/Ericsson/codechecker/blob/master/docs/report_directory.md#metadata-structure + def _merge_analyzer_statistics(stat1, stat2): """ Merges two analyzer_statistics dicts into stat1.