-
Notifications
You must be signed in to change notification settings - Fork 15.4k
KAFKA-12980: Return empty record batch from Consumer::poll when position advances due to aborted transactions #11046
New issue
Have a question about this project? Sign up for a free GitHub account to open an issue and contact its maintainers and the community.
By clicking “Sign up for GitHub”, you agree to our terms of service and privacy statement. We’ll occasionally send you account related emails.
Already on GitHub? Sign in to your account
Changes from 3 commits
4175df1
62706bd
c90d85f
e3fcdaf
cb30115
File filter
Filter by extension
Conversations
Jump to
Diff view
Diff view
There are no files selected for viewing
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -0,0 +1,123 @@ | ||
| /* | ||
| * Licensed to the Apache Software Foundation (ASF) under one or more | ||
| * contributor license agreements. See the NOTICE file distributed with | ||
| * this work for additional information regarding copyright ownership. | ||
| * The ASF licenses this file to You under the Apache License, Version 2.0 | ||
| * (the "License"); you may not use this file except in compliance with | ||
| * the License. You may obtain a copy of the License at | ||
| * | ||
| * http://www.apache.org/licenses/LICENSE-2.0 | ||
| * | ||
| * Unless required by applicable law or agreed to in writing, software | ||
| * distributed under the License is distributed on an "AS IS" BASIS, | ||
| * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. | ||
| * See the License for the specific language governing permissions and | ||
| * limitations under the License. | ||
| */ | ||
| package org.apache.kafka.clients.consumer.internals; | ||
|
|
||
| import org.apache.kafka.clients.consumer.ConsumerRecord; | ||
| import org.apache.kafka.common.TopicPartition; | ||
|
|
||
| import java.util.ArrayList; | ||
| import java.util.Collections; | ||
| import java.util.HashMap; | ||
| import java.util.List; | ||
| import java.util.Map; | ||
| import java.util.Objects; | ||
|
|
||
| import static org.apache.kafka.common.utils.Utils.mkEntry; | ||
| import static org.apache.kafka.common.utils.Utils.mkMap; | ||
|
|
||
| public class Fetch<K, V> { | ||
| private final Map<TopicPartition, List<ConsumerRecord<K, V>>> records; | ||
| private boolean positionAdvanced; | ||
| private int numRecords; | ||
|
|
||
| public static <K, V> Fetch<K, V> empty() { | ||
| return new Fetch<>(new HashMap<>(), false, 0); | ||
| } | ||
|
|
||
| public static <K, V> Fetch<K, V> forPartition( | ||
| TopicPartition partition, | ||
| List<ConsumerRecord<K, V>> records, | ||
| boolean positionAdvanced | ||
| ) { | ||
| Map<TopicPartition, List<ConsumerRecord<K, V>>> recordsMap = records.isEmpty() | ||
| ? new HashMap<>() | ||
| : mkMap(mkEntry(partition, records)); | ||
| return new Fetch<>(recordsMap, positionAdvanced, records.size()); | ||
| } | ||
|
|
||
| private Fetch( | ||
| Map<TopicPartition, List<ConsumerRecord<K, V>>> records, | ||
| boolean positionAdvanced, | ||
| int numRecords | ||
| ) { | ||
| this.records = records; | ||
| this.positionAdvanced = positionAdvanced; | ||
| this.numRecords = numRecords; | ||
| } | ||
|
|
||
| /** | ||
| * Add another {@link Fetch} to this one; all of its records will be added to this fetch's | ||
| * {@link #records()} records}, and if the other fetch | ||
| * {@link #positionAdvanced() advanced the consume position for any topic partition}, | ||
| * this fetch will be marked as having advanced the consume position as well. | ||
| * @param fetch the other fetch to add; may not be null | ||
| */ | ||
| public void add(Fetch<K, V> fetch) { | ||
| Objects.requireNonNull(fetch); | ||
| addRecords(fetch.records); | ||
| this.positionAdvanced |= fetch.positionAdvanced; | ||
| } | ||
|
|
||
| /** | ||
| * @return all of the non-control messages for this fetch, grouped by partition | ||
| */ | ||
| public Map<TopicPartition, List<ConsumerRecord<K, V>>> records() { | ||
| return Collections.unmodifiableMap(records); | ||
| } | ||
|
|
||
| /** | ||
| * @return whether the fetch caused the consumer's | ||
| * {@link org.apache.kafka.clients.consumer.KafkaConsumer#position(TopicPartition) position} to advance for at | ||
| * least one of the topic partitions in this fetch | ||
| */ | ||
| public boolean positionAdvanced() { | ||
| return positionAdvanced; | ||
| } | ||
|
|
||
| /** | ||
| * @return the total number of non-control messages for this fetch, across all partitions | ||
| */ | ||
| public int numRecords() { | ||
| return numRecords; | ||
| } | ||
|
|
||
| /** | ||
| * @return {@code true} if and only if this fetch did not return any user-visible (i.e., non-control) records, and | ||
| * did not cause the consumer position to advance for any topic partitions | ||
| */ | ||
| public boolean isEmpty() { | ||
| return numRecords == 0 && !positionAdvanced; | ||
| } | ||
|
|
||
| private void addRecords(Map<TopicPartition, List<ConsumerRecord<K, V>>> records) { | ||
| records.forEach((partition, partRecords) -> { | ||
| this.numRecords += partRecords.size(); | ||
| List<ConsumerRecord<K, V>> currentRecords = this.records.get(partition); | ||
| if (currentRecords == null) { | ||
| this.records.put(partition, partRecords); | ||
| } else { | ||
| // this case shouldn't usually happen because we only send one fetch at a time per partition, | ||
| // but it might conceivably happen in some rare cases (such as partition leader changes). | ||
| // we have to copy to a new list because the old one may be immutable | ||
| List<ConsumerRecord<K, V>> newRecords = new ArrayList<>(currentRecords.size() + partRecords.size()); | ||
| newRecords.addAll(currentRecords); | ||
| newRecords.addAll(partRecords); | ||
| this.records.put(partition, newRecords); | ||
| } | ||
| }); | ||
| } | ||
| } |
| Original file line number | Diff line number | Diff line change |
|---|---|---|
|
|
@@ -109,8 +109,6 @@ | |
| import java.util.function.Function; | ||
| import java.util.stream.Collectors; | ||
|
|
||
| import static java.util.Collections.emptyList; | ||
|
|
||
| /** | ||
| * This class manages the fetching process with the brokers. | ||
| * <p> | ||
|
|
@@ -637,15 +635,15 @@ private Map<TopicPartition, Long> beginningOrEndOffset(Collection<TopicPartition | |
| /** | ||
| * Return the fetched records, empty the record buffer and update the consumed position. | ||
| * | ||
| * NOTE: returning empty records guarantees the consumed position are NOT updated. | ||
| * NOTE: returning an {@link Fetch#isEmpty empty} fetch guarantees the consumed position is not updated. | ||
|
C0urante marked this conversation as resolved.
Outdated
|
||
| * | ||
| * @return The fetched records per partition | ||
| * @return A {@link Fetch} for the requested partitions | ||
| * @throws OffsetOutOfRangeException If there is OffsetOutOfRange error in fetchResponse and | ||
| * the defaultResetPolicy is NONE | ||
| * @throws TopicAuthorizationException If there is TopicAuthorization error in fetchResponse. | ||
| */ | ||
| public Map<TopicPartition, List<ConsumerRecord<K, V>>> fetchedRecords() { | ||
| Map<TopicPartition, List<ConsumerRecord<K, V>>> fetched = new HashMap<>(); | ||
| public Fetch<K, V> collectFetch() { | ||
| Fetch<K, V> fetch = Fetch.empty(); | ||
| Queue<CompletedFetch> pausedCompletedFetches = new ArrayDeque<>(); | ||
| int recordsRemaining = maxPollRecords; | ||
|
|
||
|
|
@@ -665,7 +663,7 @@ public Map<TopicPartition, List<ConsumerRecord<K, V>>> fetchedRecords() { | |
| // in cases such as the TopicAuthorizationException, and the second condition ensures that no | ||
| // potential data loss due to an exception in a following record. | ||
| FetchResponseData.PartitionData partition = records.partitionData; | ||
| if (fetched.isEmpty() && FetchResponse.recordsOrFail(partition).sizeInBytes() == 0) { | ||
| if (fetch.isEmpty() && FetchResponse.recordsOrFail(partition).sizeInBytes() == 0) { | ||
| completedFetches.poll(); | ||
| } | ||
| throw e; | ||
|
|
@@ -681,39 +679,24 @@ public Map<TopicPartition, List<ConsumerRecord<K, V>>> fetchedRecords() { | |
| pausedCompletedFetches.add(nextInLineFetch); | ||
| nextInLineFetch = null; | ||
| } else { | ||
| List<ConsumerRecord<K, V>> records = fetchRecords(nextInLineFetch, recordsRemaining); | ||
|
|
||
| if (!records.isEmpty()) { | ||
| TopicPartition partition = nextInLineFetch.partition; | ||
| List<ConsumerRecord<K, V>> currentRecords = fetched.get(partition); | ||
| if (currentRecords == null) { | ||
| fetched.put(partition, records); | ||
| } else { | ||
| // this case shouldn't usually happen because we only send one fetch at a time per partition, | ||
| // but it might conceivably happen in some rare cases (such as partition leader changes). | ||
| // we have to copy to a new list because the old one may be immutable | ||
| List<ConsumerRecord<K, V>> newRecords = new ArrayList<>(records.size() + currentRecords.size()); | ||
| newRecords.addAll(currentRecords); | ||
| newRecords.addAll(records); | ||
| fetched.put(partition, newRecords); | ||
| } | ||
| recordsRemaining -= records.size(); | ||
| } | ||
| Fetch<K, V> nextFetch = fetchRecords(nextInLineFetch, recordsRemaining); | ||
| recordsRemaining -= nextFetch.numRecords(); | ||
| fetch.add(nextFetch); | ||
| } | ||
| } | ||
| } catch (KafkaException e) { | ||
| if (fetched.isEmpty()) | ||
| if (fetch.isEmpty()) | ||
| throw e; | ||
| } finally { | ||
| // add any polled completed fetches for paused partitions back to the completed fetches queue to be | ||
| // re-evaluated in the next poll | ||
| completedFetches.addAll(pausedCompletedFetches); | ||
| } | ||
|
|
||
| return fetched; | ||
| return fetch; | ||
| } | ||
|
|
||
| private List<ConsumerRecord<K, V>> fetchRecords(CompletedFetch completedFetch, int maxRecords) { | ||
| private Fetch<K, V> fetchRecords(CompletedFetch completedFetch, int maxRecords) { | ||
| if (!subscriptions.isAssigned(completedFetch.partition)) { | ||
| // this can happen when a rebalance happened before fetched records are returned to the consumer's poll call | ||
| log.debug("Not returning fetched records for partition {} since it is no longer assigned", | ||
|
|
@@ -735,13 +718,21 @@ private List<ConsumerRecord<K, V>> fetchRecords(CompletedFetch completedFetch, i | |
| log.trace("Returning {} fetched records at offset {} for assigned partition {}", | ||
| partRecords.size(), position, completedFetch.partition); | ||
|
|
||
| boolean positionAdvanced = false; | ||
|
|
||
| if (completedFetch.nextFetchOffset > position.offset) { | ||
| FetchPosition nextPosition = new FetchPosition( | ||
| completedFetch.nextFetchOffset, | ||
| completedFetch.lastEpoch, | ||
| position.currentLeader); | ||
| log.trace("Update fetching position to {} for partition {}", nextPosition, completedFetch.partition); | ||
| log.trace("Updating fetch position from {} to {} for partition {} and returning {} records from `poll()`", | ||
| position, nextPosition, completedFetch.partition, partRecords.size()); | ||
| subscriptions.position(completedFetch.partition, nextPosition); | ||
| positionAdvanced = true; | ||
| if (partRecords.isEmpty()) { | ||
| log.trace("Returning empty records from `poll()` " | ||
| + "since the consumer's position has advanced for at least one topic partition"); | ||
|
Contributor
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. nit: I think this comment made more sense in its original location in
Contributor
Author
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. This is true. I was hoping we could have something in here that explicitly states that this can happen because of the change in behavior implemented in this PR (i.e., skipping control records or aborted transactions). If you think it's worth it to call that out in a log message we can do that here, otherwise the entire
Contributor
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. I'm inclined to either remove the log line entirely or move it back to its former location in
Contributor
Author
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. Moved it to |
||
| } | ||
| } | ||
|
|
||
| Long partitionLag = subscriptions.partitionLag(completedFetch.partition, isolationLevel); | ||
|
|
@@ -753,7 +744,7 @@ private List<ConsumerRecord<K, V>> fetchRecords(CompletedFetch completedFetch, i | |
| this.sensors.recordPartitionLead(completedFetch.partition, lead); | ||
| } | ||
|
|
||
| return partRecords; | ||
| return Fetch.forPartition(completedFetch.partition, partRecords, positionAdvanced); | ||
| } else { | ||
| // these records aren't next in line based on the last consumed position, ignore them | ||
| // they must be from an obsolete request | ||
|
|
@@ -765,7 +756,7 @@ private List<ConsumerRecord<K, V>> fetchRecords(CompletedFetch completedFetch, i | |
| log.trace("Draining fetched records for partition {}", completedFetch.partition); | ||
| completedFetch.drain(); | ||
|
|
||
| return emptyList(); | ||
| return Fetch.empty(); | ||
| } | ||
|
|
||
| // Visible for testing | ||
|
|
||
Uh oh!
There was an error while loading. Please reload this page.