You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何用PySpark和Graph API准确计算Microsoft Secure Score?分数不符排查

问题:通过Microsoft Graph API获取的Microsoft Secure Score与Defender显示不一致

我正尝试通过Microsoft Graph API获取Microsoft Secure Score数据,但获取到的分数和Microsoft Defender界面显示的不一致。

以下是我用PySpark写的初始代码:

import requests
import json

tenant_id = ""
client_id = ""
client_secret = ""

token_url = f"https://login.microsoftonline.com/{tenant_id}/oauth2/v2.0/token"

token_data = {
    "grant_type": "client_credentials",
    "client_id": client_id,
    "client_secret": client_secret,
    "scope": "https://graph.microsoft.com/.default"
}

response = requests.post(token_url, data=token_data)
if response.status_code == 200:
    access_token = response.json().get("access_token")
else:
    raise Exception(f"Error to obtain the token: {response.text}")

api_url = "https://graph.microsoft.com/v1.0/security/secureScores"

headers = {
    "Authorization": f"Bearer {access_token}",
    "Content-Type": "application/json"
}

api_response = requests.get(api_url, headers=headers)
if api_response.status_code == 200:
    data = api_response.json().get("value", [])
else:
    raise Exception(f"Error to call API: {api_response.text}")

if data:
    df = spark.createDataFrame(data)
    # df.show()
else:
    print("no data from API.")

for field in df.schema.fields:
    if isinstance(field.dataType, (MapType, ArrayType)):
        df = df.withColumn(field.name, to_json(col(field.name)))


df_1 = df.withColumn(
    "controlScores",
    from_json(
        col("controlScores"),
        ArrayType(
            StructType([ 
                StructField("controlCategory", StringType(), True),
                StructField("lastSynced", StringType(), True),
                StructField("score", StringType(), True),
                StructField("implementationStatus", StringType(), True),
                StructField("controlName", StringType(), True),
                StructField("description", StringType(), True),
                StructField("scoreInPercentage", StringType(), True)
            ])
        )
    )
)

df_2 = df_1.withColumn("controlScore", explode("controlScores"))

df_3 = df_2.select(
    'createdDateTime',
    'currentScore',
    'maxScore',
    col("controlScore.lastSynced").alias("lastSynced"),
    col("controlScore.controlCategory").alias("controlCategory"),
    col("controlScore.controlName").alias("controlName"),
    col("controlScore.description").alias("description"),
    col("controlScore.scoreInPercentage").alias("scoreInPercentage")
)

display(
    df_3
    .agg(avg(col('currentScore') / col('maxScore')) * 100)
)

display(df_3.groupBy('controlCategory').agg(avg('scoreInPercentage')))

运行后得到的分数和Defender里显示的完全对不上。我查了Microsoft文档里关于如何通过Graph API计算Identity SecureScore的文章,照着实现后还是不行——不管是总分、分类分数,还是通过averageComparativeScores获取的同行业对比数值,都和界面显示不一致。

我期望得到和Defender界面一致的分数结果。

更新:
我修改了代码,现在能得到Data分类的正确数值,但其他分类的分数还是和界面差异很大:

import requests
import json

tenant_id = ""
client_id = ""
client_secret = ""

token_url = f"https://login.microsoftonline.com/{tenant_id}/oauth2/v2.0/token"

token_data = {
    "grant_type": "client_credentials",
    "client_id": client_id,
    "client_secret": client_secret,
    "scope": "https://graph.microsoft.com/.default"
}

response = requests.post(token_url, data=token_data)
if response.status_code == 200:
    access_token = response.json().get("access_token")
else:
    raise Exception(f"Erro ao obter token: {response.text}")

api_url = "https://graph.microsoft.com/v1.0/security/secureScores"

headers = {
    "Authorization": f"Bearer {access_token}",
    "Content-Type": "application/json"
}

api_response = requests.get(api_url, headers=headers)
if api_response.status_code == 200:
    data = api_response.json().get("value", [])
else:
    raise Exception(f"Erro ao chamar API: {api_response.text}")

if data:
    df = spark.createDataFrame(data)
    # df.show()
else:
    print("Nenhum dado retornado pela API.")

for field in df.schema.fields:
    if isinstance(field.dataType, (MapType, ArrayType)):
        df = df.withColumn(field.name, to_json(col(field.name)))


df_1 = df.withColumn(
    "controlScores",
    from_json(
        col("controlScores"),
        ArrayType(
            StructType([
                StructField("controlCategory", StringType(), True),
                StructField("id", StringType(), True),
                StructField("lastSynced", StringType(), True),
                StructField("score", StringType(), True),
                StructField("implementationStatus", StringType(), True),
                StructField("controlName", StringType(), True),
                StructField("description", StringType(), True),
                StructField("scoreInPercentage", StringType(), True)
            ])
        )
    )
)

df_2 = df_1.withColumn("controlScore", explode("controlScores"))

df_3 = (df_2.select(
    col('createdDateTime'),
    col('currentScore'),
    col('maxScore'),
    col("controlScore.lastSynced").alias("lastSynced"),
    col("controlScore.controlCategory").alias("controlCategory"),
    col("controlScore.controlName").alias("controlName"),
    col("controlScore.description").alias("description"),
    col("controlScore.scoreInPercentage").alias("scoreInPercentage"),
    col("controlScore.id").alias("id"),
    col("controlScore.score").alias("score")
        ))

results = {}

for category in categories:
    max_score_total = 0
    score_total = 0

    control_category = df_3.filter(df_3["controlCategory"] == category).collect()

    for control in control_category:
        score = float(control["score"]) if control["score"] is not None else 0.0
        score_in_percentage = float(control["scoreInPercentage"]) if control["scoreInPercentage"] is not None else 0.0

        if score_in_percentage == 0:

            control_id = control["id"]
            
            control_profile_url = f"https://graph.microsoft.com/v1.0/security/secureScoreControlProfiles/{control_id}"
            control_profile_response = requests.get(control_profile_url, headers=headers)

            if control_profile_response.status_code == 200:
                control_profile_data = control_profile_response.json()
                max_score = control_profile_data.get("maxScore", 1) 
            else:
                max_score = 1
        else:
            max_score = score / (score_in_percentage * 0.01)

        max_score_total += max_score
        score_total += score

    per_category = (score_total / max_score_total) * 100 if max_score_total > 0 else 0

    results[category] = {
        "Score Total": score_total,
        "Max Score Total": max_score_total,
        "Percentual": per_category,
    }

print(json.dumps(results, indent=4))

有没有人遇到过类似问题?麻烦帮忙找出问题所在,任何建议或指导都非常感谢!

内容的提问来源于stack exchange,提问作者coding

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.14 21:19:54