Scroll and retrieve tophits aggregation documents using composite key

Viewed 43

I am struggling (a lot) to figure out how to write a TopHits aggregation but with scrolling through all documents, this question although tagged as C# it's not entirely dedicated to the NEST (or OpenSearch Client) .NET SDKs. See below the requirements.

  • Find the latest document for each user ID.

I tried changing the below query to get the latest document for every bucket instead of the doc_count using CompositeAggregation to retrieve the documents count for a query (similar to scrolling).

{
    "track_total_hits": false,
    "aggs": {
        "completions_users": {
            "composite": {
                "after": {
                    "influencerId": ""
                },
                "size": 10000,
                "sources": [
                    {
                        "influencerId": {
                            "terms": {
                                "field": "influencerId"
                            }
                        }
                    }
                ]
            }
        }
    },
    "query": {
        "bool": {
            "must": [
                {
                    "bool": {
                        "must": [
                            {
                                "terms": {
                                    "influencerId": [
                                        "XXXXX-ad84-4f35-8a58-9ee3cc8a3c6b",
                                        "YYYYYY-ad84-4f35-8a58-9ee3cc8a3c6b"
                                    ]
                                }
                            },
                            {
                                "match": {
                                    "campaignSponsorshipId": {
                                        "query": "XXXXXX-729e-4663-85f2-6ff3f986e93f"
                                    }
                                }
                            },
                            {
                                "match": {
                                    "status": {
                                        "query": "Completed"
                                    }
                                }
                            }
                        ]
                    }
                }
            ]
        }
    },
    "size": 0
}

And this is working amazingly great to retrieve a result like the below.

{
    "took": 11,
    "timed_out": false,
    "_shards": {
        "total": 3,
        "successful": 3,
        "skipped": 0,
        "failed": 0
    },
    "hits": {
        "max_score": null,
        "hits": []
    },
    "aggregations": {
        "completions_users": {
            "after_key": {
                "influencerId": "XXXXXX-ad84-4f35-8a58-9ee3cc8a3c6b"
            },
            "buckets": [
                {
                    "key": {
                        "influencerId": "XXXXXX-ad84-4f35-8a58-9ee3cc8a3c6b"
                    },
                    "doc_count": 6
                }
            ]
        }
    }

What I am looking for is to retrieve for each bucket the top hits similar to the below query which will just return 1 document.

{
    "track_total_hits": false,
    "aggs": {
        "search_last_completed": {
            "composite": {
                "after": {
                    "influencerId": ""
                },
                "size": 10000,
                "sources": [
                    {
                        "influencerId": {
                            "terms": {
                                "field": "influencerId"
                            }
                        }
                    }
                ]
            }
        },
        "most_recent_doc": {
            "top_hits": {
                "size": 1,
                "sort": [
                    {
                        "completedDate": {
                            "order": "desc"
                        }
                    }
                ],
                "_source": {
                    "includes": [
                        "completedDate",
                        "id",
                        "influencerId",
                        "campaignId",
                        "campaignSponsorshipSetId",
                        "campaignSponsorshipId"
                    ]
                }
            }
        }
    },
    "query": {
        "bool": {
            "must": [
                {
                    "bool": {
                        "must": [
                            {
                                "terms": {
                                    "influencerId": [
                                        "XXXXXX-85a2-40fa-9c88-f165f4685b73",
                                        "YYYYYY-85a2-40fa-9c88-f165f4685b73"
                                    ]
                                }
                            },
                            {
                                "match": {
                                    "status": {
                                        "query": "Completed"
                                    }
                                }
                            }
                        ]
                    }
                }
            ]
        }
    },
    "size": 0
}

The below response is not acceptable because I just get 1 document in response to all the composite buckets returned.

{
    "took": 16,
    "timed_out": false,
    "_shards": {
        "total": 3,
        "successful": 3,
        "skipped": 0,
        "failed": 0
    },
    "hits": {
        "max_score": null,
        "hits": []
    },
    "aggregations": {
        "most_recent_doc": { // 1 document result here
            "hits": {
                "total": {
                    "value": 99,
                    "relation": "eq"
                },
                "max_score": null,
                "hits": [
                    {
                        "_index": "sponsorshipsinfluencers-v7-2022-8",
                        "_type": "_doc",
                        "_id": "a1ad8a13-eb82-4d9c-bd8b-de9ea03c6199",
                        "_score": null,
                        "_source": {
                            "campaignSponsorshipSetId": "XXXXXXX-c57a-487e-89b9-4d787c2dc778",
                            "influencerId": "XXXXXXX-85a2-40fa-9c88-f165f4685b73",
                            "campaignId": "XXXXX-d985-4aa7-bd18-e07e5988bb0a",
                            "campaignSponsorshipId": "XXXX-729e-4663-85f2-6ff3f986e93f",
                            "id": "XXXXX-eb82-4d9c-bd8b-de9ea03c6199",
                            "completedDate": "2022-08-08T12:03:52.9172233Z"
                        },
                        "sort": [
                            1659960232917
                        ]
                    }
                ]
            }
        },
        "search_last_completed": {
            "after_key": {
                "influencerId": "XXXXXX-85a2-40fa-9c88-f165f4685b73"
            },
            "buckets": [ // Should have more info with tophits for each bucket record
                {
                    "key": {
                        "influencerId": "XXXXX-85a2-40fa-9c88-f165f4685b73"
                    },
                    "doc_count": 99
                }
            ]
        }
    }
}

From the knowledge I have so far I can't seem to find how a nested aggregation would work to have something like the below (assumption schema response).

{
  "took": 16,
  "timed_out": false,
  "_shards": {
    "total": 3,
    "successful": 3,
    "skipped": 0,
    "failed": 0
  },
  "hits": {
    "max_score": null,
    "hits": [
      
    ]
  },
  "aggregations": {
    "search_last_completed": {
      "after_key": {
        "influencerId": "XXXXXXX-85a2-40fa-9c88-f165f4685b73"
      },
      "buckets": [
        {
          "key": {
            "campaignSponsorshipSetId": "49ab4c80-c57a-487e-89b9-4d787c2dc778",
            "influencerId": "XXXXXXXX-85a2-40fa-9c88-f165f4685b73",
            "campaignId": "910330b8-d985-4aa7-bd18-e07e5988bb0a",
            "campaignSponsorshipId": "47d2fc07-729e-4663-85f2-6ff3f986e93f",
            "id": "a1ad8a13-eb82-4d9c-bd8b-de9ea03c6199",
            "completedDate": "2022-08-08T12:03:52.9172233Z"
          },
          "doc_count": 99
        }
      ]
    }
  }
}
0 Answers
Related