#include #include #include #include #include "imdb_search.h" #define MAX_ENTRIES 15 #define NUM_THREADS 16 #define PAGE_SIZE 6 // ANSI color codes #define RESET "\033[0m" #define RED "\033[31m" #define GREEN "\033[32m" #define YELLOW "\033[33m" #define BLUE "\033[34m" #define MAGENTA "\033[35m" #define CYAN "\033[36m" #define WHITE "\033[37m" #define PRINT_HEADER(title) \ printf(MAGENTA "\n━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\n" RESET); \ printf(CYAN " %s\n" RESET, title); \ printf(MAGENTA "┅┅┅┅┅┅┅┅┅┅┅┅┅┅┅┅┅┅┅┅┅┅┅┅┅┅┅┅┅┅┅┅┅┅┅┅┅┅┅┅┅\n" RESET); #define PRINT_HEADER_SMALL(title) \ printf(MAGENTA "\n–––––––––––––––––––––––––––––––––––––––––\n" RESET); \ printf(CYAN " %s\n" RESET, title); \ void trim_whitespace(char *str) { char *start = str; char *end; // Trim leading whitespace while (isspace((unsigned char)*start)) { start++; } // Shift the trimmed string to the beginning if necessary if (start != str) { memmove(str, start, strlen(start) + 1); } // Find the end of the string end = str + strlen(str) - 1; // Trim trailing whitespace while (end > str && isspace((unsigned char)*end)) { *end = '\0'; end--; } } // Utility function to convert string to lowercase void to_lowercase(char *p) { for ( ; *p; ++p) *p = tolower(*p); } // Utility function to duplicate strings safely char *strdup_(const char *s) { size_t len = strlen(s) + 1; char *dup = (char *)malloc(len); if (dup) { memcpy(dup, s, len); } return dup; } #define MAX_FIELDS 8 // Maximum number of fields in the TSV line #define FIELD_BUFFER 256 // Buffer size for individual fields // Function to split a line into fields based on tab delimiters void split_line(const char *line, char fields[MAX_FIELDS][FIELD_BUFFER], int *field_count) { char *token; char buffer[strlen(line) + 1]; strcpy(buffer, line); *field_count = 0; token = strtok(buffer, "\t"); while (token != NULL && *field_count < MAX_FIELDS) { strncpy(fields[(*field_count)++], token, FIELD_BUFFER - 1); fields[*field_count - 1][FIELD_BUFFER - 1] = '\0'; // Ensure null termination token = strtok(NULL, "\t"); } } // Function to parse a line and create a TitleLanguages struct TitleLanguages *parse_title_languages(const char *line) { char fields[MAX_FIELDS][FIELD_BUFFER]; int field_count = 0; split_line(line, fields, &field_count); if (field_count < 8) { fprintf(stderr, "Error: Invalid line format\n"); return NULL; } TitleLanguages *titleLang = malloc(sizeof(TitleLanguages)); if (!titleLang) { fprintf(stderr, "Error: Memory allocation failed\n"); return NULL; } // Parse titleId strcpy(titleLang->titleId, fields[0]); // Parse language (handle \N for missing values) titleLang->language = strdup_(fields[4]); trim_whitespace(titleLang->language); // Parse region (handle \N for missing values) titleLang->region = strdup_(fields[3]); trim_whitespace(titleLang->region); // Parse title (handle \N for missing values) titleLang->title = strdup_(fields[2]); trim_whitespace(titleLang->title); // Parse isOriginalTitle titleLang->isOriginalTitle = atoi(fields[7]); return titleLang; } // Function to load languages, regions, and titles from the file int load_languages_and_regions(TitleEntry *titles, int title_count) { if (title_count <= 0) { fprintf(stderr, "Error: Invalid title count\n"); return -1; } const char *akas_filename = "imdb/title.akas.tsv"; FILE *akas_file = fopen(akas_filename, "r"); if (!akas_file) { perror("Failed to open title.akas.tsv file"); return -1; } char line[2048]; // Skip the header line in the akas file if (fgets(line, sizeof(line), akas_file) == NULL) { fprintf(stderr, "Error: Empty file or missing header\n"); fclose(akas_file); return -1; } int tc = 0, c = 0; // Title counter while (fgets(line, sizeof(line), akas_file)) { TitleLanguages *titleLang = parse_title_languages(line); if (!titleLang) { fprintf(stderr, "Error parsing line: %s", line); continue; } // Find matching title while (tc < title_count && strcmp(titles[tc].tconst, titleLang->titleId) < 0) { tc++, c++; } if (c >= title_count) { if (c == 0) free(titleLang); c = 0; continue; } TitleEntry *currentTitle = &(titles[tc]); if (!currentTitle) { fprintf(stderr, "Error: Null TitleEntry at index %d\n", tc); free(titleLang); break; } // Allocate or reallocate memory for languages if (currentTitle->languageCount == 0) { currentTitle->languages = malloc(sizeof(TitleLanguages) * 2); if (!currentTitle->languages) { fprintf(stderr, "Error: Memory allocation failed\n"); free(titleLang); fclose(akas_file); return -1; } currentTitle->languageCount = 0; currentTitle->languageCount_max = 2; } else if (currentTitle->languageCount >= currentTitle->languageCount_max) { currentTitle->languageCount_max *= 2; TitleLanguages *new_langs = realloc(currentTitle->languages, sizeof(TitleLanguages) * currentTitle->languageCount_max); if (!new_langs) { fprintf(stderr, "Error: Memory reallocation failed\n"); free(titleLang); fclose(akas_file); return -1; } currentTitle->languages = new_langs; } // Add the new language entry currentTitle->languages[currentTitle->languageCount++] = *titleLang; free(titleLang); } fclose(akas_file); return 0; } // Load titles from TSV file // Load titles from TSV file int load_titles(TitleEntry **titles, int *title_count, RatingInfo* ratings, int ratings_count) { const char *filename = "imdb/title.basics.tsv"; FILE *file = fopen(filename, "r"); if (!file) { perror("Failed to open file"); return -1; } char line[1024]; int count = 0, capacity = 100; *titles = malloc(capacity * sizeof(TitleEntry)); // Skip the header line fgets(line, sizeof(line), file); while (fgets(line, sizeof(line), file)) { if (count >= capacity) { capacity *= 2; *titles = realloc(*titles, capacity * sizeof(TitleEntry)); } TitleEntry *entry = &(*titles)[count]; char *token = strtok(line, "\t"); strcpy(entry->tconst, token); entry->tconst[12] = '\0'; token = strtok(NULL, "\t"); strcpy(entry->titleType, token); token = strtok(NULL, "\t"); strcpy(entry->primaryTitle, token); token = strtok(NULL, "\t"); strcpy(entry->originalTitle, token); token = strtok(NULL, "\t"); entry->isAdult = atoi(token); token = strtok(NULL, "\t"); entry->startYear = atoi(token); token = strtok(NULL, "\t"); entry->endYear = atoi(token); token = strtok(NULL, "\t"); strcpy(entry->runtimeMinutesgenres, token ? token : ""); entry->languageCount = 0; entry->languageCount_max = 0; entry->ratingInfo = get_rating_by_tconst(entry->tconst, ratings, &ratings_count); count++; } fclose(file); *title_count = count; return 0; } // Comparison function for qsort int compare_results(const void *a, const void *b) { SearchResult *result_a = (SearchResult *)a; SearchResult *result_b = (SearchResult *)b; return result_b->score - result_a->score; // Sort in descending order of score } // Macro for calculating word value based on index #define CALCULATE_WORD_VALUE(index) ((index <= 3) ? 100 : (index > 12 ? 5 : 100 - (95 * (index - 3) / 9))) int string_difference_percentage(const char *query, const char *title) { int totalScore = 0; int wordCount = 0; // Count of query words int matchCount = 0; // Count of matched words const char *wordStart = query; const char *wordEnd = query; // Go through each word in the query while (*wordEnd != '\0') { // Skip to the end of the current word while (*wordEnd != ' ' && *wordEnd != '\0') { wordEnd++; } // Create a temporary substring for the word size_t wordLength = wordEnd - wordStart; char tempWord[wordLength + 1]; strncpy(tempWord, wordStart, wordLength); tempWord[wordLength] = '\0'; // Null terminate the word // Look for the word in the title if (strstr(title, tempWord)) { matchCount++; totalScore += CALCULATE_WORD_VALUE(matchCount); } // Move to the next word if (*wordEnd != '\0') { wordEnd++; // Skip the space } wordStart = wordEnd; wordCount++; } // Normalize the score: proportional to query words return wordCount > 0 ? (totalScore / wordCount) : 0; } float string_difference_percentage_(const char *query, const char *str2) { int len1 = strlen(query); // Length of the query string int len2 = strlen(str2); // Length of the second string int diff_count = 0; // Compare character by character up to the length of the query string for (int i = 0; i < len1 && i < len2; i++) { if (query[i] != str2[i]) { diff_count++; } } // Add the remaining characters from the longer string as differences (only for str2) diff_count += abs(len1 - len2); // Avoid division by zero (if the query string is empty) if (len1 == 0) { return 100.0; } // Calculate the similarity percentage based on the query string's length float similarity = 100.0 - ((float)diff_count / len1) * 100; // Ensure that similarity percentage stays within the range [0, 100] if (similarity < 0.0) { similarity = 1.0; // No similarity if completely different } else if (similarity > 100.0) { similarity = 100.0; // Full similarity if identical } return similarity; } // Function to calculate the score based on the query and title float calculate_score_single(const char *query, char* title, float rank) { float score = 0; char title_copy[66], query_copy[66];// str_part[156] = "\0"; // Words that trigger an early break if their index is greater than 3 // const char *break_words[] = {"by", "|", NULL}; // Copy and convert both title and query to lowercase for case-insensitive matching strncpy(title_copy, title, 65); strncpy(query_copy, query, 65); to_lowercase(title_copy); to_lowercase(query_copy); score = (string_difference_percentage(query_copy, title_copy) + string_difference_percentage_(query_copy, title_copy) ) / 2; //if (score == 100) score += rank; return score; } float calculate_score(const char *query, TitleEntry* entry, float rank, int* selected_id, char* selected) { float score = calculate_score_single(query, entry->originalTitle, rank); float score_n = 0; for (int i = 0; i < entry->languageCount; i++) { if (entry->languages == NULL) break; if (entry->languages[i].title == NULL) continue; // Check if the title is valid based on the given criteria if (entry->languages[i].isOriginalTitle || (entry->languages[i].region != NULL && strcmp(entry->languages[i].region, "DE") == 0) || (entry->languages[i].language != NULL && strcmp(entry->languages[i].language, "de") == 0)) { score_n = calculate_score_single(query, entry->languages[i].title, rank); if (score_n > score) { score = score_n; *selected_id = i; // Ensure selected is always assigned if (entry->languages[i].isOriginalTitle) { strcpy(selected, "original"); } else if (entry->languages[i].language != NULL && strcmp(entry->languages[i].language, "de") == 0) { strcpy(selected, "DE"); } else { strcpy(selected, "de"); } } } } return score; } #include #include #include #include // Structure for passing data to each thread typedef struct { const char *query; TitleEntry *titles; int start_idx; int end_idx; SearchResult *temp_results; int *result_count; int thread_id; int num_threads; float rank; } ThreadData; #include #include #include #include pthread_mutex_t result_mutex = PTHREAD_MUTEX_INITIALIZER; // Thread function to process a portion of titles void *search_titles_thread(void *arg) { ThreadData *data = (ThreadData *)arg; int local_result_count = 0; for (int i = data->start_idx; i < data->end_idx; i++) { float score = calculate_score(data->query, &data->titles[i], data->titles[i].ratingInfo ? data->titles[i].ratingInfo->averageRating : 0, &data->temp_results[local_result_count].selected_id, data->temp_results[local_result_count].selected); // Store the result strncpy(data->temp_results[local_result_count].tconst, data->titles[i].tconst, sizeof(data->temp_results[local_result_count].tconst) - 1); data->temp_results[local_result_count].tconst[sizeof(data->temp_results[local_result_count].tconst) - 1] = '\0'; // Ensure null termination strncpy(data->temp_results[local_result_count].primaryTitle, data->titles[i].primaryTitle, sizeof(data->temp_results[local_result_count].primaryTitle) - 1); data->temp_results[local_result_count].primaryTitle[sizeof(data->temp_results[local_result_count].primaryTitle) - 1] = '\0'; // Ensure null termination data->temp_results[local_result_count].title = &data->titles[i]; data->temp_results[local_result_count].score = score; local_result_count++; } // Safely update shared result count pthread_mutex_lock(&result_mutex); *data->result_count += local_result_count; pthread_mutex_unlock(&result_mutex); pthread_exit(NULL); } // Main search function int search_titles(const char *query, TitleEntry *titles, int title_count, SearchResult *results, int max_results) { int num_threads = NUM_THREADS; // Adjust based on available cores pthread_t threads[num_threads]; ThreadData thread_data[num_threads]; int result_count = 0; SearchResult *temp_results = malloc(title_count * sizeof(SearchResult)); // Divide titles among threads int chunk_size = title_count / num_threads; for (int i = 0; i < num_threads; i++) { thread_data[i].query = query; thread_data[i].titles = titles; thread_data[i].start_idx = i * chunk_size; thread_data[i].end_idx = (i == num_threads - 1) ? title_count : (i + 1) * chunk_size; thread_data[i].temp_results = temp_results + (i * chunk_size); thread_data[i].result_count = &result_count; thread_data[i].thread_id = i; thread_data[i].num_threads = num_threads; pthread_create(&threads[i], NULL, search_titles_thread, (void *)&thread_data[i]); } // Wait for threads to finish for (int i = 0; i < num_threads; i++) { pthread_join(threads[i], NULL); } // Sort results by score (descending order) qsort(temp_results, result_count, sizeof(SearchResult), compare_results); // Copy top results int limit = result_count < max_results ? result_count : max_results; for (int i = 0; i < limit; i++) { results[i] = temp_results[i]; } free(temp_results); return limit; } SearchResult* select_result(SearchResult *results, int found, const char* search_query) { int choice; int page = 0; int just_movie = 0; // Create an array to store the filtered movie results SearchResult* filtered_results[found]; int filtered_found = 0; while (1) { // Filter results if just_movie is enabled if (just_movie) { filtered_found = 0; for (int i = 0; i < found; i++) { if (strcmp(results[i].title->titleType, "movie") == 0) { filtered_results[filtered_found] = &results[i]; // Store movie result address filtered_found++; } } } else { // If not filtering for movies, show all results filtered_found = found; for (int i = 0; i < found; i++) { filtered_results[i] = &results[i]; // Store all result addresses } } int page_size = PAGE_SIZE; int pages = (filtered_found + page_size - 1) / page_size; // Calculate number of pages printf(MAGENTA "\n┏━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\n" RESET); printf(MAGENTA "┃ " WHITE "%-24s" CYAN " (%d/%d)\n" RESET, search_query, page + 1, pages); printf(MAGENTA "┠━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\n" RESET); // Display results for the current page int displayed = 0; for (int i = page * page_size; i < (page + 1) * page_size && displayed < page_size && i < filtered_found; i++) { SearchResult *result = filtered_results[i]; // Display basic result info with colors printf(MAGENTA "┃" GREEN "%2d.%s %s%-24s%s %s(Score: %.2f)%s\n", i + 1, RESET, YELLOW, result->primaryTitle, RESET, CYAN, result->score, RESET); if (strlen(result->selected) > 0) printf(MAGENTA "┃" RESET " %s: " WHITE "%-24s\n" RESET, result->selected, result->title->languages[result->selected_id].title); // Get rating and vote count float r = result->title->ratingInfo != NULL ? result->title->ratingInfo->averageRating : 0; int rc = result->title->ratingInfo != NULL ? result->title->ratingInfo->numVotes : 0; // Display detailed information with color printf(MAGENTA "┃" RESET " Type: " WHITE "%s, " RESET "Rank: " CYAN "%.2f, " RESET "Rank_c: " CYAN "%d " YELLOW "%s\n" RESET, result->title->titleType, r, rc, result->selected); // Display year, ID, and genres with colors printf(MAGENTA "┃" RESET " Year: " WHITE "%d-%d, " RESET "ID: " WHITE "%s, " RESET "Genres: " WHITE "%s\n" RESET, result->title->startYear, result->title->endYear, result->title->tconst, result->title->runtimeMinutesgenres); fflush(stdout); displayed++; } printf(MAGENTA "┠──────────────────────────────────────────────\n" RESET); printf(MAGENTA "┃ " WHITE "%-24s" CYAN " (%d/%d)\n" RESET, search_query, page + 1, pages); printf(MAGENTA "┗━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━\n" RESET); printf("\nSelect (%s1-%d%s), [n|p] next/prev page, [q|movie]:\n", YELLOW, filtered_found, RESET); char input[10]; fgets(input, sizeof(input), stdin); trim_whitespace(input); if (input[0] == 'q') { // Return an empty result or handle quit as needed return NULL; } else if (strcmp(input, "movie") == 0) { just_movie = !just_movie; } else if (input[0] == 'n' && page < pages - 1) { page++; // Go to the next page } else if (input[0] == 'p' && page > 0) { page--; // Go to the previous page } else { // Ensure we are reading a valid number and not an out-of-bounds selection if (sscanf(input, "%d", &choice) == 1 && choice >= 1 && choice <= filtered_found) { return filtered_results[choice - 1]; // Return the selected result } else { printf("Invalid selection, try again.\n"); } } } } // Function to search and select a title based on a query SearchResult* search_and_select(const char *query, TitleEntry *titles, int title_count) { printf("\n\n\n–––––––––––––––––––––––––––––––––––––––––––––––––––––––\n"); printf("Search and select\n"); printf("Title: %s\n", query); SearchResult results[1000]; // Increase size if needed SearchResult* selected = NULL; int found = search_titles(query, titles, title_count, results, 1000); printf("Found %d results:\n", found); if (found > 0) { printf("########################################\n"); printf("### Results for: %s\n", query); // Call select_result to allow the user to select a result selected = select_result(results, found, query); if (selected == NULL) return NULL; // Check if the result is empty or the user chose to quit if (selected->score == 0 && selected->primaryTitle[0] == '\0') { printf("\nNo selection was made.\n"); } // Print the selected result printf("\nYou selected: %s (Score: %.2f)\n", selected->title->primaryTitle, selected->score); printf("Year: %d, Genres: %s\n", selected->title->startYear, selected->title->runtimeMinutesgenres); } else { printf("No results found.\n"); } return selected; // Exit the function early }