mirror of
				https://github.com/MariaDB/server.git
				synced 2025-10-25 18:38:00 +03:00 
			
		
		
		
	We implement an idea that was suggested by Michael 'Monty' Widenius in October 2017: When InnoDB is inserting into an empty table or partition, we can write a single undo log record TRX_UNDO_EMPTY, which will cause ROLLBACK to clear the table. For this to work, the insert into an empty table or partition must be covered by an exclusive table lock that will be held until the transaction has been committed or rolled back, or the INSERT operation has been rolled back (and the table is empty again), in lock_table_x_unlock(). Clustered index records that are covered by the TRX_UNDO_EMPTY record will carry DB_TRX_ID=0 and DB_ROLL_PTR=1<<55, and thus they cannot be distinguished from what MDEV-12288 leaves behind after purging the history of row-logged operations. Concurrent non-locking reads must be adjusted: If the read view was created before the INSERT into an empty table, then we must continue to imagine that the table is empty, and not try to read any records. If the read view was created after the INSERT was committed, then all records must be visible normally. To implement this, we introduce the field dict_table_t::bulk_trx_id. This special handling only applies to the very first INSERT statement of a transaction for the empty table or partition. If a subsequent statement in the transaction is modifying the initially empty table again, we must enable row-level undo logging, so that we will be able to roll back to the start of the statement in case of an error (such as duplicate key). INSERT IGNORE will continue to use row-level logging and locking, because implementing it would require the ability to roll back the latest row. Since the undo log that we write only allows us to roll back the entire statement, we cannot support INSERT IGNORE. We will introduce a handler::extra() parameter HA_EXTRA_IGNORE_INSERT to indicate to storage engines that INSERT IGNORE is being executed. In many test cases, we add an extra record to the table, so that during the 'interesting' part of the test, row-level locking and logging will be used. Replicas will continue to use row-level logging and locking until MDEV-24622 has been addressed. Likewise, this optimization will be disabled in Galera cluster until MDEV-24623 enables it. dict_table_t::bulk_trx_id: The latest active or committed transaction that initiated an insert into an empty table or partition. Protected by exclusive table lock and a clustered index leaf page latch. ins_node_t::bulk_insert: Whether bulk insert was initiated. trx_t::mod_tables: Use C++11 style accessors (emplace instead of insert). Unlike earlier, this collection will cover also temporary tables. trx_mod_table_time_t: Add start_bulk_insert(), end_bulk_insert(), is_bulk_insert(), was_bulk_insert(). trx_undo_report_row_operation(): Before accessing any undo log pages, invoke trx->mod_tables.emplace() in order to determine whether undo logging was disabled, or whether this is the first INSERT and we are supposed to write a TRX_UNDO_EMPTY record. row_ins_clust_index_entry_low(): If we are inserting into an empty clustered index leaf page, set the ins_node_t::bulk_insert flag for the subsequent trx_undo_report_row_operation() call. lock_rec_insert_check_and_lock(), lock_prdt_insert_check_and_lock(): Remove the redundant parameter 'flags' that can be checked in the caller. btr_cur_ins_lock_and_undo(): Simplify the logic. Correctly write DB_TRX_ID,DB_ROLL_PTR after invoking trx_undo_report_row_operation(). trx_mark_sql_stat_end(), ha_innobase::extra(HA_EXTRA_IGNORE_INSERT), ha_innobase::external_lock(): Invoke trx_t::end_bulk_insert() so that the next statement will not be covered by table-level undo logging. ReadView::changes_visible(trx_id_t) const: New accessor for the case where the trx_id_t is not read from a potentially corrupted index page but directly from the memory. In this case, we can skip a sanity check. row_sel(), row_sel_try_search_shortcut(), row_search_mvcc(): row_sel_try_search_shortcut_for_mysql(), row_merge_read_clustered_index(): Check dict_table_t::bulk_trx_id. row_sel_clust_sees(): Replaces lock_clust_rec_cons_read_sees(). lock_sec_rec_cons_read_sees(): Replaced with lower-level code. btr_root_page_init(): Refactored from btr_create(). dict_index_t::clear(), dict_table_t::clear(): Empty an index or table, for the ROLLBACK of an INSERT operation. ROW_T_EMPTY, ROW_OP_EMPTY: Note a concurrent ROLLBACK of an INSERT into an empty table. This is joint work with Thirunarayanan Balathandayuthapani, who created a working prototype. Thanks to Matthias Leich for extensive testing.
		
			
				
	
	
		
			121 lines
		
	
	
		
			3.9 KiB
		
	
	
	
		
			Plaintext
		
	
	
	
	
	
			
		
		
	
	
			121 lines
		
	
	
		
			3.9 KiB
		
	
	
	
		
			Plaintext
		
	
	
	
	
	
| --source include/have_innodb.inc
 | |
| --source include/have_debug_sync.inc
 | |
| --source include/have_log_bin.inc
 | |
| 
 | |
| # Test some group commit code paths by using debug_sync to do controlled
 | |
| # commits of 6 transactions: first 1 alone, then 3 as a group, then 2 as a
 | |
| # group.
 | |
| #
 | |
| # Group 3 is allowed to race as far as possible ahead before group 2 finishes
 | |
| # to check some edge case for concurrency control.
 | |
| 
 | |
| CREATE TABLE t1 (a VARCHAR(10) PRIMARY KEY) ENGINE=innodb;
 | |
| # MDEV-515 takes X-LOCK on the table for the first insert.
 | |
| # So concurrent insert won't happen on the table
 | |
| INSERT INTO t1 VALUES("default");
 | |
| 
 | |
| SELECT variable_value INTO @commits FROM information_schema.global_status
 | |
|  WHERE variable_name = 'binlog_commits';
 | |
| SELECT variable_value INTO @group_commits FROM information_schema.global_status
 | |
|  WHERE variable_name = 'binlog_group_commits';
 | |
| 
 | |
| connect(con1,localhost,root,,);
 | |
| connect(con2,localhost,root,,);
 | |
| connect(con3,localhost,root,,);
 | |
| connect(con4,localhost,root,,);
 | |
| connect(con5,localhost,root,,);
 | |
| connect(con6,localhost,root,,);
 | |
| 
 | |
| # Start group1 (with one thread) doing commit, waiting for
 | |
| # group2 to queue up before finishing.
 | |
| 
 | |
| connection con1;
 | |
| SET DEBUG_SYNC= "commit_before_get_LOCK_after_binlog_sync SIGNAL group1_running WAIT_FOR group2_queued";
 | |
| send INSERT INTO t1 VALUES ("con1");
 | |
| 
 | |
| # Make group2 (with three threads) queue up.
 | |
| # Make sure con2 is the group commit leader for group2.
 | |
| # Make group2 wait with running commit_ordered() until group3 has committed.
 | |
| 
 | |
| connection con2;
 | |
| set DEBUG_SYNC= "now WAIT_FOR group1_running";
 | |
| SET DEBUG_SYNC= "commit_after_prepare_ordered SIGNAL group2_con2";
 | |
| SET DEBUG_SYNC= "commit_before_get_LOCK_after_binlog_sync SIGNAL group2_running";
 | |
| SET DEBUG_SYNC= "commit_after_release_LOCK_log WAIT_FOR group3_committed";
 | |
| send INSERT INTO t1 VALUES ("con2");
 | |
| connection con3;
 | |
| SET DEBUG_SYNC= "now WAIT_FOR group2_con2";
 | |
| SET DEBUG_SYNC= "commit_after_prepare_ordered SIGNAL group2_con3";
 | |
| send INSERT INTO t1 VALUES ("con3");
 | |
| connection con4;
 | |
| SET DEBUG_SYNC= "now WAIT_FOR group2_con3";
 | |
| SET DEBUG_SYNC= "commit_after_prepare_ordered SIGNAL group2_con4";
 | |
| SET DEBUG_SYNC= "commit_after_group_run_commit_ordered SIGNAL group2_visible WAIT_FOR group2_checked";
 | |
| send INSERT INTO t1 VALUES ("con4");
 | |
| 
 | |
| # When group2 is queued, let group1 continue and queue group3.
 | |
| 
 | |
| connection default;
 | |
| SET DEBUG_SYNC= "now WAIT_FOR group2_con4";
 | |
| 
 | |
| # At this point, trasaction 1 is still not visible as commit_ordered() has not
 | |
| # been called yet.
 | |
| SET SESSION TRANSACTION ISOLATION LEVEL READ COMMITTED;
 | |
| SELECT * FROM t1 ORDER BY a;
 | |
| 
 | |
| SET DEBUG_SYNC= "now SIGNAL group2_queued";
 | |
| connection con1;
 | |
| reap;
 | |
| 
 | |
| # Now transaction 1 is visible.
 | |
| connection default;
 | |
| SELECT * FROM t1 ORDER BY a;
 | |
| 
 | |
| connection con5;
 | |
| SET DEBUG_SYNC= "commit_before_get_LOCK_after_binlog_sync SIGNAL group3_con5";
 | |
| SET DEBUG_SYNC= "commit_after_get_LOCK_log SIGNAL con5_leader WAIT_FOR con6_queued";
 | |
| set DEBUG_SYNC= "now WAIT_FOR group2_running";
 | |
| send INSERT INTO t1 VALUES ("con5");
 | |
| 
 | |
| connection con6;
 | |
| SET DEBUG_SYNC= "now WAIT_FOR con5_leader";
 | |
| SET DEBUG_SYNC= "commit_after_prepare_ordered SIGNAL con6_queued";
 | |
| send INSERT INTO t1 VALUES ("con6");
 | |
| 
 | |
| connection default;
 | |
| SET DEBUG_SYNC= "now WAIT_FOR group3_con5";
 | |
| # Still only transaction 1 visible, as group2 have not yet run commit_ordered().
 | |
| SELECT * FROM t1 ORDER BY a;
 | |
| SET DEBUG_SYNC= "now SIGNAL group3_committed";
 | |
| SET DEBUG_SYNC= "now WAIT_FOR group2_visible";
 | |
| # Now transactions 1-4 visible.
 | |
| SELECT * FROM t1 ORDER BY a;
 | |
| SET DEBUG_SYNC= "now SIGNAL group2_checked";
 | |
| 
 | |
| connection con2;
 | |
| reap;
 | |
| 
 | |
| connection con3;
 | |
| reap;
 | |
| 
 | |
| connection con4;
 | |
| reap;
 | |
| 
 | |
| connection con5;
 | |
| reap;
 | |
| 
 | |
| connection con6;
 | |
| reap;
 | |
| 
 | |
| connection default;
 | |
| # Check all transactions finally visible.
 | |
| SELECT * FROM t1 ORDER BY a;
 | |
| 
 | |
| SELECT variable_value - @commits FROM information_schema.global_status
 | |
|  WHERE variable_name = 'binlog_commits';
 | |
| SELECT variable_value - @group_commits FROM information_schema.global_status
 | |
|  WHERE variable_name = 'binlog_group_commits';
 | |
| 
 | |
| SET DEBUG_SYNC= 'RESET';
 | |
| DROP TABLE t1;
 |