Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 7 additions & 0 deletions src/BUILD
Original file line number Diff line number Diff line change
Expand Up @@ -20,6 +20,13 @@ py_binary(
visibility = ["//visibility:public"],
)

py_binary(
name = "metadata_merge",
srcs = ["metadata_merge.py"],
# Bazel 6 won't see this otherwise
visibility = ["//visibility:public"],
)

py_binary(
name = "codechecker_script",
srcs = ["codechecker_script.py"],
Expand Down
199 changes: 199 additions & 0 deletions src/metadata_merge.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,199 @@
# Copyright 2023 Ericsson AB
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.

"""
Merges the metadata information of multiple CodeChecker analysis.
"""

import json
import os
import sys
from typing import List, Dict, Any


# Structure of metadata files is defined here:
# https://github.com/Ericsson/codechecker/blob/master/docs/report_directory.md#metadata-structure

def _merge_analyzer_statistics(stat1, stat2):
"""
Merges two analyzer_statistics dicts into stat1.

Sums failed/successful counts, extends source lists,
and preserves the version field.
"""
stat1["failed"] = stat1["failed"] + stat2["failed"]
stat1["failed_sources"].extend(stat2["failed_sources"])
stat1["successful"] = stat1["successful"] + stat2["successful"]
stat1["successful_sources"].extend(stat2["successful_sources"])


def _merge_checkers(checkers1, checkers2):
"""
Merges two checker dicts.

A checker is enabled (True) in the result if it is enabled
in either input. This handles the case where different
per-file runs may report slightly different checker sets.
"""
for checker, enabled in checkers2.items():
if checker not in checkers1:
checkers1[checker] = enabled
else:
# If enabled in either, mark as enabled
checkers1[checker] = checkers1[checker] or enabled


def _merge_analyzers(analyzers1, analyzers2):
"""
Merges analyzer sections from two metadata files.

Handles the case where json2 has analyzers not present
in json1 by initializing them from json2.
"""
for analyzer_name, analyzer_data in analyzers2.items():
if analyzer_name not in analyzers1:
# Analyzer exists in json2 but not json1; adopt it
analyzers1[analyzer_name] = analyzer_data
continue

# Merge checkers
if "checkers" in analyzer_data:
if "checkers" not in analyzers1[analyzer_name]:
analyzers1[analyzer_name]["checkers"] = {}
_merge_checkers(
analyzers1[analyzer_name]["checkers"],
analyzer_data["checkers"],
)

# Merge analyzer_statistics
if "analyzer_statistics" in analyzer_data:
if "analyzer_statistics" not in analyzers1[analyzer_name]:
analyzers1[analyzer_name]["analyzer_statistics"] = (
analyzer_data["analyzer_statistics"]
)
else:
_merge_analyzer_statistics(
analyzers1[analyzer_name]["analyzer_statistics"],
analyzer_data["analyzer_statistics"],
)


def merge_two_json(json1, json2):
"""
Merges the data of two metadata json files.

If both json files are empty, returns an empty json.
Handles all fields defined in the CodeChecker metadata spec:
- version, name, action_num, command, working_directory,
output_path, result_source_files, analyzers (with checkers
and analyzer_statistics including version), skipped,
timestamps.
"""
if json1 == {}:
return json2
if json2 == {}:
return json1
# Happens when analysis of all files was skipped
if json1 == {} and json2 == {}:
return {}
# Fail if the metadata format version is not 2
assert json1["version"] == 2
assert json2["version"] == 2
json1_tools = json1["tools"][0]
json2_tools = json2["tools"][0]
# We expect the following fields to be the same in all
# metadata files from the same analysis invocation.
assert json1_tools["name"] == json2_tools["name"]
# Same CodeChecker version
assert json1_tools["version"] == json2_tools["version"]
# command, working_directory and output_path may differ
# between per-file runs (e.g. remote workers). We keep
# json1's values as the canonical ones.

# Sum action counts and skipped files
json1_tools["action_num"] += json2_tools["action_num"]
json1_tools["skipped"] = json1_tools["skipped"] + json2_tools["skipped"]

# Merge result_source_files mapping
json1_tools["result_source_files"].update(
json2_tools["result_source_files"]
)

# Merge timestamps; we assume both json files describe jobs
# in the same analysis invocation, implying that the analysis
# start time is the lowest timestamp, and the end is the
# highest.
# Note: caching will break this assumption

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

That sounds like a bigger deal than this Note implies.

Copy link
Copy Markdown
Contributor Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I extended the Note, not sure what else could I do here.

# Users may see months, or even years long difference in timestamps
json1_tools["timestamps"]["begin"] = min(
float(json1_tools["timestamps"]["begin"]),
float(json2_tools["timestamps"]["begin"]),
)
json1_tools["timestamps"]["end"] = max(
float(json1_tools["timestamps"]["end"]),
float(json2_tools["timestamps"]["end"]),
)

# Merge analyzers (checkers + analyzer_statistics)
_merge_analyzers(
json1_tools["analyzers"], json2_tools["analyzers"]
)

return json1


def merge_json_files(file_paths: List[str]) -> Dict[str, Any]:
"""
Merges a list of metadata.json files, using merge_two_json.

Returns the merged contents of the json files.
"""
merged_data = {}
for file_path in file_paths:
if not os.path.exists(file_path):
print(
f"Error: File not found at '{file_path}'. Skipping.",
file=sys.stderr,
)
continue

try:
with open(file_path, "r", encoding="utf-8") as f:
data = json.load(f)
merged_data = merge_two_json(merged_data, data)
except json.JSONDecodeError:
print(
f"Error: Could not decode JSON from '{file_path}'. Skipping.",
file=sys.stderr,
)

return merged_data

def main():
"""
Main function of metadata merge
"""
output_file = sys.argv[1]
input_files = sys.argv[2:]

merged_data = merge_json_files(input_files)
if merged_data:
with open(output_file, "w", encoding="utf-8") as f:
json.dump(merged_data, f, indent=4)
else:
print("\nNo data was merged. Output file will not be created.")


if __name__ == "__main__":
main()
34 changes: 30 additions & 4 deletions src/per_file.bzl
Original file line number Diff line number Diff line change
Expand Up @@ -90,7 +90,6 @@ def _run_code_checker(
clang_tidy_plist,
clangsa_plist,
codechecker_log,
codechecker_metadata,
]

analyzer_output_paths = "clangsa," + clangsa_plist.path + \
Expand All @@ -105,7 +104,9 @@ def _run_code_checker(
# TODO: use env for environment variables, instead of passing it as argument
ctx.actions.run(
inputs = inputs,
outputs = outputs,
# We do not want all individual metadata files
# cluttering the data folder
outputs = outputs + [codechecker_metadata],
executable = per_file_script,
tools = [
info.runfiles,
Expand Down Expand Up @@ -137,7 +138,7 @@ def _run_code_checker(
mnemonic = "CodeChecker",
progress_message = "CodeChecker analyze {}".format(src.short_path),
)
return outputs
return outputs, codechecker_metadata

def check_valid_file_type(src):
"""
Expand Down Expand Up @@ -180,6 +181,23 @@ def _collect_all_sources_and_headers(ctx):
all_files += headers
return all_files

def _merge_metadata(ctx, metadata):
"""
Merges metadata files of individual CodeChecker runs into 1

Returns the metadata file objects
"""
metadata_json = ctx.actions.declare_file(ctx.attr.name + "/data/metadata.json")
ctx.actions.run(
inputs = metadata,
outputs = [metadata_json],
executable = ctx.executable._metadata_merge,
arguments = [metadata_json.path] + [file.path for file in metadata],
mnemonic = "Metadata",
progress_message = "Merging metadata.json",
)
return metadata_json

def _per_file_impl(ctx):
info = ctx.toolchains["//:toolchain_type"].codecheckerinfo
compile_commands = None
Expand All @@ -192,6 +210,7 @@ def _per_file_impl(ctx):
fail("Seems compile_commands.json file is incorrect!")
sources_and_headers = _collect_all_sources_and_headers(ctx)
options = ctx.attr.default_options + ctx.attr.options
all_metadata = []
config_file, env_vars = get_config_file(ctx)
all_files = [compile_commands, config_file]

Expand All @@ -213,7 +232,7 @@ def _per_file_impl(ctx):
if not check_valid_file_type(src):
continue
args = target[SourceFilesInfo].compilation_db.to_list()
outputs = _run_code_checker(
outputs, metadata = _run_code_checker(
ctx,
per_file_script,
src,
Expand All @@ -229,6 +248,8 @@ def _per_file_impl(ctx):
sources_and_headers,
)
all_files += outputs
all_metadata.append(metadata)
all_files.append(_merge_metadata(ctx, all_metadata))
ctx.actions.write(
output = ctx.outputs.test_script,
is_executable = True,
Expand Down Expand Up @@ -296,6 +317,11 @@ per_file_test = rule(
"When set, tools from this target are used instead of " +
"Bazel's toolchain resolution.",
),
"_metadata_merge": attr.label(
default = ":metadata_merge",
executable = True,
cfg = "exec",
),
"_per_file_script": attr.label(
executable = True,
cfg = "exec",
Expand Down
5 changes: 2 additions & 3 deletions test/unit/metadata/BUILD
Original file line number Diff line number Diff line change
Expand Up @@ -61,10 +61,9 @@ unit_test(

unit_test(
name = "per_file_metadata_test",
contains = [r"\"action_num\": 1,"],
contains = [r"\"action_num\": 2,"],
data = [":per_file_multiple_source"],
files = [
"test/unit/metadata/per_file_multiple_source/data/test-unit-metadata-main.cc_metadata.json",
"test/unit/metadata/per_file_multiple_source/data/test-unit-metadata-other.cc_metadata.json",
"test/unit/metadata/per_file_multiple_source/data/metadata.json",
],
)
Loading