import pandas as pd # Sample DataFrame data = { 'predicted_categories': [ ['80814001 - Freze Uçları', '13003106 - Freze', '80805004 - Sanayi Makineleri'], ['13003144 - Torna Makinesi', '13003195 - Kumpas'] ], 'pred_category_id': [80814001, 13003195], 'text_predicted_probs': [ [0.943, 0.018, 0.008], [0.6, 0.4] ] } df = pd.DataFrame(data) # 1. Explode both list columns simultaneously to maintain alignment between category and probability df_exploded = df.explode(['predicted_categories', 'text_predicted_probs']) # 2. Extract the numeric ID from the category string using vectorized regex df_exploded['extracted_id'] = df_exploded['predicted_categories'].str.extract(r'^(\d+)').astype(float) # 3. Filter for rows where the extracted ID matches the target 'pred_category_id' matched = df_exploded[df_exploded['extracted_id'] == df_exploded['pred_category_id']] # 4. Dedup the index (safety net in case an ID appears twice within the same list) matched = matched[~matched.index.duplicated(keep='first')] # 5. Map the extracted probability column back to the original DataFrame using the index df['pred_category_prob'] = matched['text_predicted_probs'] df